From 02b104028028ada0c980c76dbf30f27cdbb27f84 Mon Sep 17 00:00:00 2001 From: Paul Bakaus Date: Sun, 12 Jul 2026 17:55:10 -0700 Subject: [PATCH] Add Live performance lab and benchmarks Measure framework and provider latency, enforce fidelity and cleanup gates, and publish reproducible results on the dev-only Live Lab.\n\nAI-assisted: OpenAI Codex. --- package.json | 1 + scripts/benchmark-live-control.mjs | 62 ++ scripts/benchmark-live-init.mjs | 102 ++++ scripts/benchmark-live-providers.mjs | 539 +++++++++++++++++ scripts/benchmark-live.mjs | 299 ++++++++++ scripts/compare-live-benchmarks.mjs | 48 ++ scripts/lib/live-benchmark.mjs | 395 +++++++++++++ scripts/lib/live-provider-benchmark.mjs | 547 ++++++++++++++++++ site/data/live-annotated-result.ts | 20 + site/data/live-control-result.ts | 17 + site/data/live-framework-results.ts | 29 + site/data/live-harnesses.ts | 144 +++-- site/data/live-init-result.ts | 14 + site/data/live-progressive-result.ts | 25 + site/data/live-provider-result.ts | 78 +++ site/pages/live-lab/index.astro | 181 +++++- site/styles/live-performance.css | 92 +++ .../files/index.html | 12 + .../files/package.json | 18 + .../files/src/App.jsx | 26 + .../files/src/main.jsx | 10 + .../files/src/styles.css | 111 ++++ .../files/vite.config.js | 10 + .../vite8-react-brand-fidelity/fixture.json | 37 ++ .../vite8-react-brand-fidelity/gitignore.txt | 3 + tests/live-benchmark.test.mjs | 165 ++++++ tests/live-provider-benchmark.test.mjs | 115 ++++ tests/live-provider-benchmark/README.md | 54 ++ 28 files changed, 3096 insertions(+), 58 deletions(-) create mode 100644 scripts/benchmark-live-control.mjs create mode 100644 scripts/benchmark-live-init.mjs create mode 100644 scripts/benchmark-live-providers.mjs create mode 100644 scripts/benchmark-live.mjs create mode 100644 scripts/compare-live-benchmarks.mjs create mode 100644 scripts/lib/live-benchmark.mjs create mode 100644 scripts/lib/live-provider-benchmark.mjs create mode 100644 site/data/live-annotated-result.ts create mode 100644 site/data/live-control-result.ts create mode 100644 site/data/live-framework-results.ts create mode 100644 site/data/live-init-result.ts create mode 100644 site/data/live-progressive-result.ts create mode 100644 site/data/live-provider-result.ts create mode 100644 tests/framework-fixtures/vite8-react-brand-fidelity/files/index.html create mode 100644 tests/framework-fixtures/vite8-react-brand-fidelity/files/package.json create mode 100644 tests/framework-fixtures/vite8-react-brand-fidelity/files/src/App.jsx create mode 100644 tests/framework-fixtures/vite8-react-brand-fidelity/files/src/main.jsx create mode 100644 tests/framework-fixtures/vite8-react-brand-fidelity/files/src/styles.css create mode 100644 tests/framework-fixtures/vite8-react-brand-fidelity/files/vite.config.js create mode 100644 tests/framework-fixtures/vite8-react-brand-fidelity/fixture.json create mode 100644 tests/framework-fixtures/vite8-react-brand-fidelity/gitignore.txt create mode 100644 tests/live-benchmark.test.mjs create mode 100644 tests/live-provider-benchmark.test.mjs create mode 100644 tests/live-provider-benchmark/README.md diff --git a/package.json b/package.json index 62f0bbb76..da12ab1a6 100644 --- a/package.json +++ b/package.json @@ -69,6 +69,7 @@ "bench:detector": "node scripts/benchmark-detector.mjs", "bench:detector:browser": "node scripts/benchmark-detector.mjs --browser", "bench:live": "node scripts/benchmark-live.mjs", + "bench:live:providers": "node scripts/benchmark-live-providers.mjs", "audit": "bun audit --audit-level=moderate", "prepack": "cp README.md README.repo.md && cp README.npm.md README.md", "postpack": "cp README.repo.md README.md && rm README.repo.md", diff --git a/scripts/benchmark-live-control.mjs b/scripts/benchmark-live-control.mjs new file mode 100644 index 000000000..86d705cce --- /dev/null +++ b/scripts/benchmark-live-control.mjs @@ -0,0 +1,62 @@ +#!/usr/bin/env node + +import { execFileSync } from 'node:child_process'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); +const iterations = Math.max(1, Number(arg('--iterations') || 5)); +const fixture = arg('--fixture') || 'vite8-react-plain'; +const metricsFile = path.join(os.tmpdir(), 'impeccable-live-control-' + process.pid + '.jsonl'); + +try { + for (let index = 0; index < iterations; index += 1) { + execFileSync('bun', ['run', 'test:live-e2e'], { + cwd: root, + stdio: 'ignore', + timeout: 120_000, + env: { + ...process.env, + IMPECCABLE_E2E_ONLY: fixture, + IMPECCABLE_E2E_SCENARIOS: 'progressive', + IMPECCABLE_E2E_METRICS_FILE: metricsFile, + }, + }); + } + + const rows = fs.readFileSync(metricsFile, 'utf-8').trim().split('\n').filter(Boolean).map(JSON.parse); + console.log(JSON.stringify({ + fixture, + iterations: rows.length, + measuredAt: new Date().toISOString(), + acceptToPicking: summarize(rows.map((row) => row.acceptToPickingMs)), + nextGoToPickup: summarize(rows.map((row) => row.nextGoToPickupMs)), + samples: rows, + }, null, 2)); +} finally { + try { fs.unlinkSync(metricsFile); } catch {} +} + +function summarize(values) { + const sorted = [...values].sort((a, b) => a - b); + return { + medianMs: percentile(sorted, 0.5), + p95Ms: percentile(sorted, 0.95), + minMs: sorted[0], + maxMs: sorted.at(-1), + }; +} + +function percentile(sorted, p) { + const index = (sorted.length - 1) * p; + const lower = Math.floor(index); + const upper = Math.ceil(index); + return Math.round((sorted[lower] * (1 - (index - lower)) + sorted[upper] * (index - lower)) * 100) / 100; +} + +function arg(name) { + const index = process.argv.indexOf(name); + return index >= 0 ? process.argv[index + 1] : undefined; +} diff --git a/scripts/benchmark-live-init.mjs b/scripts/benchmark-live-init.mjs new file mode 100644 index 000000000..5998b887d --- /dev/null +++ b/scripts/benchmark-live-init.mjs @@ -0,0 +1,102 @@ +#!/usr/bin/env node + +import { execFileSync } from 'node:child_process'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); +const liveScript = path.join(root, 'skill/scripts/live.mjs'); +const serverScript = path.join(root, 'skill/scripts/live-server.mjs'); +const iterations = Math.max(1, Number(arg('--iterations') || 10)); +const fixture = arg('--fixture') || 'vite8-react-plain'; +const fixtureDir = path.join(root, 'tests/framework-fixtures', fixture, 'files'); +const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'impeccable-live-init-')); + +try { + fs.cpSync(fixtureDir, tmp, { recursive: true }); + fs.writeFileSync(path.join(tmp, 'PRODUCT.md'), '# Product\n\nA realistic Live initialization benchmark fixture.\n'); + fs.writeFileSync(path.join(tmp, 'DESIGN.md'), '# Design\n\nUse the fixture\'s existing type, color, and component system.\n'); + fs.mkdirSync(path.join(tmp, '.impeccable/live'), { recursive: true }); + fs.writeFileSync(path.join(tmp, '.impeccable/live/config.json'), JSON.stringify({ + files: ['index.html'], + insertBefore: '', + commentSyntax: 'html', + cspChecked: true, + }, null, 2) + '\n'); + + const cold = []; + for (let i = 0; i < iterations; i += 1) { + stop(); + cold.push(runLive()); + } + + stop(); + runLive(); + const warm = []; + for (let i = 0; i < iterations; i += 1) warm.push(runLive()); + + console.log(JSON.stringify({ + fixture, + iterations, + measuredAt: new Date().toISOString(), + cold: summarize(cold), + warm: summarize(warm), + samples: { cold, warm }, + }, null, 2)); +} finally { + stop(); + fs.rmSync(tmp, { recursive: true, force: true }); +} + +function runLive() { + const start = performance.now(); + const stdout = execFileSync(process.execPath, [liveScript], { + cwd: tmp, + encoding: 'utf-8', + timeout: 15_000, + }); + const elapsed = performance.now() - start; + const result = JSON.parse(stdout); + if (!result.ok) throw new Error('live init failed: ' + stdout); + return round(elapsed); +} + +function stop() { + try { + execFileSync(process.execPath, [serverScript, 'stop'], { + cwd: tmp, + stdio: 'ignore', + timeout: 5_000, + }); + } catch {} +} + +function summarize(samples) { + const sorted = [...samples].sort((a, b) => a - b); + return { + medianMs: percentile(sorted, 0.5), + p95Ms: percentile(sorted, 0.95), + minMs: sorted[0], + maxMs: sorted.at(-1), + }; +} + +function percentile(sorted, value) { + if (sorted.length === 1) return sorted[0]; + const index = (sorted.length - 1) * value; + const lower = Math.floor(index); + const upper = Math.ceil(index); + const weight = index - lower; + return round(sorted[lower] * (1 - weight) + sorted[upper] * weight); +} + +function round(value) { + return Math.round(value * 100) / 100; +} + +function arg(name) { + const index = process.argv.indexOf(name); + return index >= 0 ? process.argv[index + 1] : undefined; +} diff --git a/scripts/benchmark-live-providers.mjs b/scripts/benchmark-live-providers.mjs new file mode 100644 index 000000000..274348c06 --- /dev/null +++ b/scripts/benchmark-live-providers.mjs @@ -0,0 +1,539 @@ +#!/usr/bin/env node + +import { execFile } from 'node:child_process'; +import { readFile, writeFile, mkdir } from 'node:fs/promises'; +import { dirname, join, resolve } from 'node:path'; +import { performance } from 'node:perf_hooks'; +import { promisify } from 'node:util'; +import { fileURLToPath } from 'node:url'; + +import { bootFixtureSession, FIXTURES_DIR } from '../tests/live-e2e/session.mjs'; +import { createFakeAgent } from '../tests/live-e2e/agent.mjs'; +import { + clickAccept, + clickGo, + pickElement, + waitForCycling, + waitForHandshake, +} from '../tests/live-e2e/ui.mjs'; +import { + BRAND_CONTRACT, + PROVIDER_PROFILES, + STRATEGIES, + applyRuntimeSourceScore, + createProviderLiveAgent, + loadBenchmarkEnv, + resolveProviderSelection, + scoreVariantOutput, + summarizeProviderRuns, + validateAcceptedCleanup, +} from './lib/live-provider-benchmark.mjs'; + +const execFileP = promisify(execFile); +const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..'); +const FIXTURE_NAME = 'vite8-react-brand-fidelity'; +const SOURCE_FILE = 'src/App.jsx'; +const args = parseArgs(process.argv.slice(2)); +const iterations = positiveInt(args.iterations, 1); +const providers = csv(args.providers || 'anthropic,openai,google'); +const strategies = csv(args.strategies || Object.keys(STRATEGIES).join(',')); +const outputPath = args.output ? resolve(ROOT, String(args.output)) : null; +const loadedEnv = loadBenchmarkEnv({ repoRoot: ROOT, explicitPath: args.envFile ? resolve(String(args.envFile)) : null }); +const modelOverrides = Object.fromEntries(providers.map((provider) => [provider, args[`${provider}Model`]]).filter(([, value]) => value)); +const selection = resolveProviderSelection(providers, modelOverrides); +const fixture = JSON.parse(await readFile(join(FIXTURES_DIR, FIXTURE_NAME, 'fixture.json'), 'utf-8')); +const liveSpec = await readFile(join(ROOT, 'skill', 'reference', 'live.md'), 'utf-8'); + +validateConfiguration({ fixture, strategies, selection, liveSpec }); + +if (args.dryRun) { + const report = { + schemaVersion: 1, + generatedAt: new Date().toISOString(), + mode: 'dry-run', + fixture: FIXTURE_NAME, + iterations, + envFilesLoaded: loadedEnv.length, + providers: selection.map(publicProviderSelection), + strategies: strategies.map((strategy) => ({ strategy, ...STRATEGIES[strategy] })), + plannedApiCallsPerIteration: Object.fromEntries(strategies.map((strategy) => [strategy, callsPerStrategy(strategy)])), + qualityGate: qualityGateDescription(), + }; + if (outputPath) await persist(report, outputPath); + process.stdout.write(JSON.stringify(report, null, 2) + '\n'); + process.exit(0); +} + +const available = selection.filter((item) => item.keyPresent); +if (args.requireAll && available.length !== selection.length) { + const missing = selection.filter((item) => !item.keyPresent).map((item) => item.provider); + throw new Error(`missing API keys for: ${missing.join(', ')}`); +} +if (available.length === 0) throw new Error('no provider API keys found; use --dry-run to validate without network calls'); + +const needsBrowser = args.pipeline === 'e2e' || args.skipCleanupControl !== true; +const { chromium } = needsBrowser ? await import('playwright') : { chromium: null }; +const browser = chromium ? await chromium.launch({ headless: args.headed !== true }) : null; +const results = []; +let cleanupControl = { passed: true, skipped: true }; +try { + if (args.skipCleanupControl !== true) { + process.stderr.write('[live-provider-bench] running provider-independent Accept/cleanup control\n'); + cleanupControl = await runCleanupControl({ browser, fixture }); + } + if (args.cleanupOnly) { + const cleanupReport = { + schemaVersion: 1, + generatedAt: new Date().toISOString(), + mode: 'cleanup-control', + fixture: FIXTURE_NAME, + cleanupControl, + }; + if (outputPath) await persist(cleanupReport, outputPath); + process.stdout.write(JSON.stringify(cleanupReport, null, 2) + '\n'); + process.exitCode = cleanupControl.passed ? 0 : 1; + } + if (args.cleanupOnly) { + // Skip provider calls; the finally block still closes Chromium. + } else { + for (const providerConfig of available) { + for (const strategy of strategies) { + for (let iteration = 1; iteration <= iterations; iteration += 1) { + process.stderr.write(`[live-provider-bench] ${providerConfig.provider}/${providerConfig.model} ${strategy} run ${iteration}/${iterations}\n`); + results.push(args.pipeline === 'e2e' + ? await runOne({ browser, fixture, liveSpec, providerConfig, strategy, iteration }) + : await runGenerationOne({ liveSpec, providerConfig, strategy, iteration, cleanupControl })); + } + } + } + } +} finally { + if (browser) await browser.close().catch(() => {}); +} + +if (args.cleanupOnly) process.exit(process.exitCode || 0); + +const groups = []; +for (const providerConfig of selection) { + for (const strategy of strategies) { + const runs = results.filter((run) => run.provider === providerConfig.provider && run.strategy === strategy); + if (runs.length === 0) continue; + groups.push({ + provider: providerConfig.provider, + model: providerConfig.model, + strategy, + strategyConfig: STRATEGIES[strategy], + summary: summarizeProviderRuns(runs), + runs, + }); + } +} + +const report = { + schemaVersion: 1, + generatedAt: new Date().toISOString(), + mode: 'live', + fixture: FIXTURE_NAME, + iterations, + providers: selection.map(publicProviderSelection), + qualityGate: qualityGateDescription(), + cleanupControl, + groups, + evaluations: evaluateStrategies(groups), + totals: { + apiCalls: results.reduce((sum, run) => sum + run.providerCalls.filter((call) => call.phase !== 'parallel-assembled').length, 0), + estimatedCostUsd: roundUsd(results.reduce((sum, run) => sum + run.estimatedCostUsd, 0)), + passingRuns: results.filter((run) => run.passed).length, + totalRuns: results.length, + }, +}; + +if (outputPath) await persist(report, outputPath); +process.stdout.write(JSON.stringify(report, null, 2) + '\n'); + +async function runGenerationOne({ liveSpec: loadedLiveSpec, providerConfig, strategy, iteration, cleanupControl: cleanup }) { + const records = []; + const agent = createProviderLiveAgent({ + provider: providerConfig.provider, + model: providerConfig.model, + strategy, + liveSpec: loadedLiveSpec, + onRecord: (record) => { + records.push(record); + const result = record.error ? `error=${record.error.split('\n')[0]}` : `duration=${record.durationMs ?? 0}ms`; + process.stderr.write(`[live-provider-bench:model] ${record.phase}${record.lane ? `/${record.lane}` : ''} attempt=${record.attempt} ${result}\n`); + }, + }); + const event = syntheticEvent(`${providerConfig.provider}-${strategy}-${iteration}`); + const startedAt = performance.now(); + try { + let output; + let firstOutput; + let firstReviewableMs; + if (typeof agent.generateFirstVariant === 'function') { + firstOutput = await agent.generateFirstVariant(event, {}); + firstReviewableMs = roundMs(performance.now() - startedAt); + output = await agent.generateRemainingVariants(event, { firstOutput }); + } else { + output = await agent.generateVariants(event, {}); + firstReviewableMs = roundMs(performance.now() - startedAt); + } + const allReadyMs = roundMs(performance.now() - startedAt); + const quality = scoreVariantOutput(output); + const estimatedCostUsd = roundUsd(records.reduce((sum, record) => sum + Number(record.estimatedCostUsd || 0), 0)); + return { + provider: providerConfig.provider, + model: providerConfig.model, + strategy, + iteration, + firstReviewableMs, + allReadyMs, + acceptCleanupMs: cleanup.acceptCleanupMs ?? null, + quality, + cleanup, + firstOutputScore: firstOutput ? scoreVariantOutput(firstOutput) : quality, + providerCalls: records.map(publicProviderRecord), + estimatedCostUsd, + passed: quality.passed && cleanup.passed, + }; + } catch (error) { + return { + provider: providerConfig.provider, + model: providerConfig.model, + strategy, + iteration, + error: String(error?.stack || error), + cleanup, + providerCalls: records.map(publicProviderRecord), + estimatedCostUsd: roundUsd(records.reduce((sum, record) => sum + Number(record.estimatedCostUsd || 0), 0)), + passed: false, + }; + } +} + +async function runCleanupControl({ browser, fixture: loadedFixture }) { + let session; + try { + session = await bootFixtureSession({ + name: FIXTURE_NAME, + fixture: loadedFixture, + browser, + agent: createFakeAgent(), + wrapTarget: { classes: 'offer-card', tag: 'article', text: 'Field Notes' }, + progressive: false, + log: args.verbose ? (message) => process.stderr.write(`[live-provider-bench:cleanup] ${message}\n`) : () => {}, + }); + await waitForHandshake(session.page); + await pickElement(session.page, loadedFixture.runtime.pickSelector); + await clickGo(session.page); + await waitForCycling(session.page, 3, { timeout: 45_000 }); + const acceptAt = performance.now(); + await clickAccept(session.page, { expectedVariant: 1 }); + const browserClean = await waitForAcceptCleanup(session.page, session.tmp); + const acceptCleanupMs = roundMs(performance.now() - acceptAt); + const source = await readFile(join(session.tmp, SOURCE_FILE), 'utf-8'); + const build = args.skipBuild ? { passed: true, skipped: true } : await verifyBuild(session.tmp); + return { + ...validateAcceptedCleanup({ source, browserClean, buildPassed: build.passed }), + acceptCleanupMs, + build, + consoleErrorCount: session.consoleErrors.length, + }; + } catch (error) { + return { passed: false, error: String(error?.stack || error) }; + } finally { + if (session) await session.teardown(); + } +} + +async function runOne({ browser, fixture, liveSpec, providerConfig, strategy, iteration }) { + const records = []; + const agent = createProviderLiveAgent({ + provider: providerConfig.provider, + model: providerConfig.model, + strategy, + liveSpec, + onRecord: (record) => { + records.push(record); + const result = record.error ? `error=${record.error.split('\n')[0]}` : `duration=${record.durationMs ?? 0}ms`; + process.stderr.write(`[live-provider-bench:model] ${record.phase}${record.lane ? `/${record.lane}` : ''} attempt=${record.attempt} ${result}\n`); + }, + }); + let session; + const startedAt = performance.now(); + try { + session = await bootFixtureSession({ + name: FIXTURE_NAME, + fixture, + browser, + agent, + wrapTarget: (event) => ({ + classes: event.element?.classes?.join(',') || 'offer-card', + tag: event.element?.tagName?.toLowerCase() || 'article', + text: event.element?.textContent?.trim(), + }), + progressive: STRATEGIES[strategy].delivery !== 'atomic', + log: args.verbose ? (message) => process.stderr.write(`[live-provider-bench:e2e] ${message}\n`) : () => {}, + }); + await waitForHandshake(session.page); + await pickElement(session.page, fixture.runtime.pickSelector); + + const goAt = performance.now(); + const firstReady = waitForFirstReviewable(session.page); + await clickGo(session.page); + await firstReady; + const firstReviewableMs = roundMs(performance.now() - goAt); + await waitForCycling(session.page, 3, { timeout: 240_000 }); + const allReadyMs = roundMs(performance.now() - goAt); + + const finalRecord = [...records].reverse().find((record) => ['atomic', 'remaining', 'parallel-assembled'].includes(record.phase) && record.output); + if (!finalRecord) throw new Error('provider benchmark produced no complete variant output'); + let quality = scoreVariantOutput(finalRecord.output); + + const acceptAt = performance.now(); + await clickAccept(session.page, { expectedVariant: 1 }); + const browserClean = await waitForAcceptCleanup(session.page, session.tmp); + const acceptCleanupMs = roundMs(performance.now() - acceptAt); + const source = await readFile(join(session.tmp, SOURCE_FILE), 'utf-8'); + const build = args.skipBuild ? { passed: true, skipped: true } : await verifyBuild(session.tmp); + const cleanup = validateAcceptedCleanup({ source, browserClean, buildPassed: build.passed }); + quality = applyRuntimeSourceScore(quality, cleanup); + + const estimatedCostUsd = roundUsd(records.reduce((sum, record) => sum + Number(record.estimatedCostUsd || 0), 0)); + const passed = quality.passed && cleanup.passed; + return { + provider: providerConfig.provider, + model: providerConfig.model, + strategy, + iteration, + firstReviewableMs, + allReadyMs, + acceptCleanupMs, + endToEndMs: roundMs(performance.now() - startedAt), + quality, + cleanup, + build, + consoleErrorCount: session.consoleErrors.length, + providerCalls: records.map(publicProviderRecord), + estimatedCostUsd, + passed, + }; + } catch (error) { + return { + provider: providerConfig.provider, + model: providerConfig.model, + strategy, + iteration, + error: String(error?.stack || error), + providerCalls: records.map(publicProviderRecord), + estimatedCostUsd: roundUsd(records.reduce((sum, record) => sum + Number(record.estimatedCostUsd || 0), 0)), + passed: false, + }; + } finally { + if (session) await session.teardown(); + } +} + +async function waitForFirstReviewable(page) { + await page.waitForFunction(() => { + const query = window.__impeccableLiveQuery || ((selector) => document.querySelector(selector)); + const wrapper = query('[data-impeccable-variants]'); + if (!wrapper) return false; + const sourceVariants = wrapper.querySelectorAll('[data-impeccable-variant]:not([data-impeccable-variant="original"])'); + const debug = window.__IMPECCABLE_LIVE_CHROME_CORE__?.debugState?.(); + const arrived = wrapper.dataset.impeccablePreview === 'svelte-component' + ? Number(debug?.arrivedVariants || 0) + : sourceVariants.length; + return arrived >= 1; + }, undefined, { timeout: 240_000 }); +} + +async function waitForAcceptCleanup(page, tmp) { + const deadline = Date.now() + 45_000; + while (Date.now() < deadline) { + const browserClean = await page.evaluate(() => { + const query = window.__impeccableLiveQuery || ((selector) => document.querySelector(selector)); + const wrapperGone = !query('[data-impeccable-variants]'); + const state = document.documentElement.dataset.impeccableLiveState; + return wrapperGone && (!state || state === 'PICKING'); + }).catch(() => false); + const source = await readFile(join(tmp, SOURCE_FILE), 'utf-8').catch(() => ''); + const sourceClean = source && !/data-impeccable-|impeccable-(?:variants|carbonize|params|original)/i.test(source); + if (browserClean && sourceClean) return true; + await new Promise((resolvePromise) => setTimeout(resolvePromise, 40)); + } + return false; +} + +async function verifyBuild(tmp) { + const startedAt = performance.now(); + try { + await execFileP('npm', ['run', 'build'], { cwd: tmp, timeout: 120_000, maxBuffer: 4 * 1024 * 1024 }); + return { passed: true, durationMs: roundMs(performance.now() - startedAt) }; + } catch (error) { + return { + passed: false, + durationMs: roundMs(performance.now() - startedAt), + error: String(error?.stderr || error?.message || error).slice(0, 2000), + }; + } +} + +function evaluateStrategies(groups) { + const evaluations = []; + for (const provider of new Set(groups.map((group) => group.provider))) { + const providerGroups = groups.filter((group) => group.provider === provider); + const baseline = providerGroups.find((group) => group.strategy === 'atomic-full'); + const baselineValid = baseline?.summary.gatePassRate === 1 + && Number.isFinite(baseline?.summary.metrics.firstReviewableMs?.median); + for (const group of providerGroups) { + const summary = group.summary; + const qualityPass = summary.gatePassRate === 1 && summary.cleanupPassRate === 1; + const first = summary.metrics.firstReviewableMs?.median; + const baselineFirst = baseline?.summary.metrics.firstReviewableMs?.median; + const firstImprovement = Number.isFinite(first) && Number.isFinite(baselineFirst) && baselineFirst > 0 + ? Number((1 - first / baselineFirst).toFixed(4)) + : null; + const latencyPass = group.strategy === 'atomic-full' + || (baselineValid ? firstImprovement != null && firstImprovement > 0.1 : Number.isFinite(first) && first < 15_000); + evaluations.push({ + provider, + model: group.model, + strategy: group.strategy, + decision: qualityPass && latencyPass ? 'accept' : 'reject', + firstReviewableImprovementVsAtomic: firstImprovement, + qualityPass, + latencyPass, + reason: !qualityPass + ? 'Rejected: fidelity, source validity, or cleanup gate failed.' + : !latencyPass + ? 'Rejected: first-reviewable median did not improve by more than 10%.' + : group.strategy === 'atomic-full' + ? 'Control: retained as the one-call baseline.' + : !baselineValid + ? 'Accepted: quality passed and first review completed under 15 seconds; the atomic control was invalid for this provider.' + : 'Accepted: materially faster first review with all quality and cleanup gates intact.', + }); + } + } + return evaluations; +} + +function publicProviderSelection(item) { + return { + provider: item.provider, + label: item.label, + model: item.model, + keyPresent: item.keyPresent, + pricePerMillion: item.pricePerMillion, + effort: item.effort, + priceSource: item.priceSource, + }; +} + +function publicProviderRecord(record) { + return { + phase: record.phase, + lane: record.lane, + attempt: record.attempt, + durationMs: record.durationMs, + totalPhaseMs: record.totalPhaseMs, + usage: record.usage, + estimatedCostUsd: record.estimatedCostUsd, + error: record.error, + outputScore: record.output ? scoreVariantOutput(record.output) : undefined, + }; +} + +function qualityGateDescription() { + return { + deterministic: true, + pass: 'overall >= 0.90 and every dimension >= 0.75; accepted source must build and contain no Live markers', + dimensions: ['brandFidelity', 'componentFidelity', 'tokenFidelity', 'copyFidelity', 'sourceValidity', 'acceptCleanup'], + identityLock: BRAND_CONTRACT.identity, + }; +} + +function syntheticEvent(id) { + const outerHTML = BRAND_CONTRACT.sourceExcerpt + .replaceAll('className=', 'class=') + .replace(/\s+/g, ' ') + .trim(); + return { + id, + action: 'impeccable', + freeformPrompt: 'Make this offer easier to scan while staying unmistakably inside the existing brand and component system.', + count: 3, + mode: 'replace', + element: { + outerHTML, + tagName: 'ARTICLE', + className: 'offer-card', + classes: ['offer-card'], + textContent: BRAND_CONTRACT.requiredCopy.join(' '), + }, + }; +} + +function callsPerStrategy(strategy) { + if (strategy === 'atomic-full') return 1; + if (strategy === 'parallel-compact') return 3; + return 2; +} + +function validateConfiguration({ fixture: loadedFixture, strategies: selectedStrategies, selection: selectedProviders, liveSpec: loadedLiveSpec }) { + if (!loadedFixture.runtime?.pickSelector) throw new Error('benchmark fixture requires runtime.pickSelector'); + if (!loadedLiveSpec.includes('Phase A: Extract the identity')) throw new Error('live.md identity-lock guidance not found'); + for (const strategy of selectedStrategies) if (!STRATEGIES[strategy]) throw new Error(`unknown strategy ${strategy}`); + if (selectedProviders.length === 0) throw new Error('at least one provider is required'); + for (const provider of selectedProviders) if (!PROVIDER_PROFILES[provider.provider]) throw new Error(`unknown provider ${provider.provider}`); +} + +async function persist(report, file) { + await mkdir(dirname(file), { recursive: true }); + await writeFile(file, JSON.stringify(report, null, 2) + '\n', 'utf-8'); + process.stderr.write(`[live-provider-bench] wrote ${file}\n`); +} + +function parseArgs(argv) { + const out = {}; + for (let position = 0; position < argv.length; position += 1) { + const arg = argv[position]; + if (!arg.startsWith('--')) continue; + const body = arg.slice(2); + const index = body.indexOf('='); + if (index !== -1) { + out[camel(body.slice(0, index))] = body.slice(index + 1); + continue; + } + const next = argv[position + 1]; + if (next !== undefined && !next.startsWith('--')) { + out[camel(body)] = next; + position += 1; + } else { + out[camel(body)] = true; + } + } + return out; +} + +function camel(value) { + return value.replace(/-([a-z])/g, (_, letter) => letter.toUpperCase()); +} + +function csv(value) { + return String(value).split(',').map((item) => item.trim()).filter(Boolean); +} + +function positiveInt(value, fallback) { + const parsed = Number.parseInt(value, 10); + return Number.isFinite(parsed) && parsed > 0 ? parsed : fallback; +} + +function roundMs(value) { + return Number(Number(value).toFixed(2)); +} + +function roundUsd(value) { + return Number(Number(value).toFixed(6)); +} diff --git a/scripts/benchmark-live.mjs b/scripts/benchmark-live.mjs new file mode 100644 index 000000000..0256f3098 --- /dev/null +++ b/scripts/benchmark-live.mjs @@ -0,0 +1,299 @@ +#!/usr/bin/env node + +import { mkdir, readFile, writeFile } from 'node:fs/promises'; +import { dirname, join, resolve } from 'node:path'; +import { fileURLToPath } from 'node:url'; + +import { createFakeAgent } from '../tests/live-e2e/agent.mjs'; +import { createLlmAgent, resolveLlmAgentConfig } from '../tests/live-e2e/agents/llm-agent.mjs'; +import { bootFixtureSession, FIXTURES_DIR } from '../tests/live-e2e/session.mjs'; +import { + clickDiscard, + clickGo, + drawAnnotationPinAndStroke, + pickElement, + waitForCycling, + waitForHandshake, +} from '../tests/live-e2e/ui.mjs'; +import { + buildInteractionRun, + assembleSplitProgressiveOutput, + createBenchmarkReport, + createTraceRecorder, + mergeBenchmarkReports, +} from './lib/live-benchmark.mjs'; + +const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..'); +const args = parseArgs(process.argv.slice(2)); +const fixtureName = String(args.fixture || 'vite8-react-plain'); +const iterations = positiveInt(args.iterations, 5); +const agentMode = args.agent === 'llm' ? 'llm' : 'fake'; +const scenario = args.scenario === 'annotated' ? 'annotated' : 'plain'; +const delivery = args.delivery === 'progressive' ? 'progressive' : 'atomic'; +const simulatedTailMs = positiveInt(args.simulatedTailMs, 0); +const outputPath = args.output ? resolve(ROOT, String(args.output)) : null; +const fixture = JSON.parse(await readFile(join(FIXTURES_DIR, fixtureName, 'fixture.json'), 'utf-8')); +if (!fixture.runtime) throw new Error(`fixture ${fixtureName} has no runtime configuration`); +if (fixture.runtime.mode === 'insert') throw new Error('live benchmark currently measures replace-mode fixtures only'); + +const { chromium } = await import('playwright'); +const browser = await chromium.launch({ headless: args.headed !== true }); +const recorder = createTraceRecorder(); +let session; + +try { + const agentInfo = await resolveAgent(agentMode, args); + if (delivery === 'progressive' && agentMode === 'llm') { + agentInfo.agent = createSplitProgressiveAgent(agentInfo.agent); + } + session = await bootFixtureSession({ + name: fixtureName, + fixture, + browser, + agent: agentInfo.agent, + wrapTarget: wrapTargetFromPickedElement, + trace: recorder.trace, + progressive: delivery === 'progressive', + progressiveDelayMs: delivery === 'progressive' ? simulatedTailMs : 0, + atomicDelayMs: delivery === 'atomic' ? simulatedTailMs : 0, + log: args.quiet ? () => {} : (message) => process.stderr.write(`[live-bench] ${message}\n`), + }); + + recorder.mark('setup.handshake.start'); + session.page.on('request', (request) => { + if (!request.url().endsWith('/events') || request.method() !== 'POST') return; + let payload; + try { payload = request.postDataJSON(); } catch { return; } + if (payload?.type === 'generate' && payload.id) { + recorder.mark('browser.generate_post', { + id: payload.id, + hasScreenshotPath: typeof payload.screenshotPath === 'string' && payload.screenshotPath.length > 0, + commentCount: Array.isArray(payload.comments) ? payload.comments.length : 0, + strokeCount: Array.isArray(payload.strokes) ? payload.strokes.length : 0, + }); + } + }); + await waitForHandshake(session.page); + recorder.mark('setup.handshake.end'); + await installBrowserTimingProbe(session.page); + + const runs = []; + const pickSelector = fixture.runtime.pickSelector || 'h1.hero-title'; + for (let iteration = 1; iteration <= iterations; iteration += 1) { + await pickElement(session.page, pickSelector, { resetPickMode: iteration > 1 }); + if (scenario === 'annotated') { + await drawAnnotationPinAndStroke(session.page, { comment: 'Benchmark annotation' }); + } + await resetBrowserTimingProbe(session.page, iteration); + + const goStarted = recorder.mark('ui.go.start', { iteration, scenario }); + const firstVariant = waitForFirstVariant(session.page).then(() => { + recorder.mark('browser.first_variant', { iteration, scenario }); + }); + + await clickGo(session.page); + recorder.mark('ui.generating_visible', { iteration, scenario }); + await firstVariant; + await waitForCycling(session.page, 3, { timeout: agentMode === 'llm' ? 150_000 : 30_000 }); + recorder.mark('browser.all_variants', { iteration, scenario }); + const browserTiming = await readBrowserTimingProbe(session.page); + + const run = buildInteractionRun(recorder.events, { + iteration, + scenario, + goStartedAt: goStarted.at, + browserTiming, + }); + assertScenarioEvidence(run, scenario); + runs.push(run); + + if (!args.quiet) process.stderr.write(formatRun(runs.at(-1)) + '\n'); + await clickDiscard(session.page); + await waitForReset(session.page); + } + + const report = createBenchmarkReport({ + fixture: fixtureName, + agent: agentMode, + provider: agentInfo.provider, + model: agentInfo.model, + scenario, + runs, + events: recorder.events, + harnessProbe: args.harnessProbe || null, + delivery, + promptMode: agentInfo.promptMode, + simulation: simulatedTailMs > 0 ? { remainingGenerationMs: simulatedTailMs } : null, + }); + + let output = report; + if (outputPath && args.append) { + try { + const existing = JSON.parse(await readFile(outputPath, 'utf-8')); + const previousReports = Array.isArray(existing.reports) ? existing.reports : [existing]; + output = mergeBenchmarkReports([...previousReports, report]); + } catch (error) { + if (error.code !== 'ENOENT') throw error; + } + } + const json = JSON.stringify(output, null, 2) + '\n'; + if (outputPath) { + await mkdir(dirname(outputPath), { recursive: true }); + await writeFile(outputPath, json, 'utf-8'); + process.stderr.write(`[live-bench] wrote ${outputPath}\n`); + } + process.stdout.write(json); +} finally { + if (session) await session.teardown(); + await browser.close().catch(() => {}); +} + +async function resolveAgent(mode, options) { + if (mode === 'fake') return { agent: createFakeAgent(), provider: 'deterministic', model: null, promptMode: null }; + const config = resolveLlmAgentConfig({ + provider: options.provider, + model: options.model, + }); + const agent = await createLlmAgent({ + config, + includeLiveSpec: false, + log: (message) => process.stderr.write(`[live-bench:llm] ${message}\n`), + }); + if (!agent) { + throw new Error(`LLM benchmark provider=${config.provider} requires ${config.requiredEnv}. Pass it in the environment; .env files are not read implicitly.`); + } + return { agent, provider: config.provider, model: config.model, promptMode: 'synthetic-element-contract' }; +} + +function createSplitProgressiveAgent(agent) { + const firstBySession = new Map(); + return { + ...agent, + async generateFirstVariant(event, context) { + const first = await agent.generateVariants({ + ...event, + count: 1, + progressive: { phase: 'first', totalCount: event.count }, + }, context); + firstBySession.set(event.id, first); + return first; + }, + async generateRemainingVariants(event, context) { + const first = firstBySession.get(event.id) || context.firstOutput; + const remaining = await agent.generateVariants({ + ...event, + count: event.count, + progressive: { + phase: 'remaining', + totalCount: event.count, + firstVariant: first?.variants?.[0] || null, + omitFirstVariantCss: true, + }, + }, context); + firstBySession.delete(event.id); + return assembleSplitProgressiveOutput(first, remaining); + }, + }; +} + +async function waitForFirstVariant(page) { + const handle = await page.waitForFunction(() => { + const wrappers = [...document.querySelectorAll('[data-impeccable-variant]')]; + return wrappers.some((element) => element.getAttribute('data-impeccable-variant') !== 'original'); + }, undefined, { timeout: 150_000 }); + await handle.dispose(); +} + +async function waitForReset(page) { + await page.waitForFunction(() => !document.querySelector('[data-impeccable-variants]'), undefined, { timeout: 30_000 }); + await page.waitForTimeout(100); +} + +async function installBrowserTimingProbe(page) { + await page.evaluate(() => { + const state = { iteration: 0, goAt: null, generateAt: null }; + window.__IMPECCABLE_LIVE_BENCH_TIMING__ = state; + const root = window.__IMPECCABLE_LIVE_CHROME_CORE__?.root?.() + || window.__IMPECCABLE_LIVE_UI_ROOT__ + || document; + root.addEventListener('click', (event) => { + const button = event.composedPath().find((node) => + node?.getAttribute?.('aria-label') === 'Generate variants' + ); + if (button) state.goAt = performance.now(); + }, true); + + const originalFetch = window.fetch.bind(window); + window.fetch = (input, init) => { + try { + const url = typeof input === 'string' ? input : input?.url; + if (String(url || '').endsWith('/events') && init?.method === 'POST') { + const payload = typeof init.body === 'string' ? JSON.parse(init.body) : null; + if (payload?.type === 'generate') state.generateAt = performance.now(); + } + } catch { /* measurement must never affect Live */ } + return originalFetch(input, init); + }; + }); +} + +async function resetBrowserTimingProbe(page, iteration) { + await page.evaluate((nextIteration) => { + const state = window.__IMPECCABLE_LIVE_BENCH_TIMING__; + if (!state) return; + state.iteration = nextIteration; + state.goAt = null; + state.generateAt = null; + }, iteration); +} + +async function readBrowserTimingProbe(page) { + return page.evaluate(() => { + const state = window.__IMPECCABLE_LIVE_BENCH_TIMING__; + return state ? { ...state } : null; + }); +} + +function assertScenarioEvidence(run, currentScenario) { + const evidence = run.annotationEvidence; + if (currentScenario === 'annotated') { + if (!evidence?.screenshotPath || evidence.comments < 1 || evidence.strokes < 1) { + throw new Error(`iteration ${run.iteration}: annotated generate payload lost screenshot/comments/strokes`); + } + return; + } + if (evidence?.screenshotPath) { + throw new Error(`iteration ${run.iteration}: plain generate payload unexpectedly included screenshotPath`); + } +} + +function wrapTargetFromPickedElement(event) { + const element = event.element || {}; + return { + elementId: element.id || undefined, + classes: Array.isArray(element.classes) ? element.classes.join(',') : undefined, + tag: element.tagName ? String(element.tagName).toLowerCase() : undefined, + text: element.textContent ? String(element.textContent).trim() : undefined, + }; +} + +function parseArgs(argv) { + const out = {}; + for (const arg of argv) { + if (!arg.startsWith('--')) continue; + const body = arg.slice(2); + const index = body.indexOf('='); + if (index === -1) out[body] = true; + else out[body.slice(0, index)] = body.slice(index + 1); + } + return out; +} + +function positiveInt(value, fallback) { + const parsed = Number.parseInt(value, 10); + return Number.isFinite(parsed) && parsed > 0 ? parsed : fallback; +} + +function formatRun(run) { + return `[live-bench] run ${run.iteration}: first=${run.goToFirstVariantMs}ms all=${run.goToAllVariantsMs}ms generation=${run.generationMs}ms overhead=${run.impeccableOverheadMs}ms`; +} diff --git a/scripts/compare-live-benchmarks.mjs b/scripts/compare-live-benchmarks.mjs new file mode 100644 index 000000000..094af5531 --- /dev/null +++ b/scripts/compare-live-benchmarks.mjs @@ -0,0 +1,48 @@ +#!/usr/bin/env node + +import { readFile } from 'node:fs/promises'; +import { resolve } from 'node:path'; + +import { compareModelBackedReports } from './lib/live-benchmark.mjs'; + +const args = parseArgs(process.argv.slice(2)); +if (!args.atomic || !args.progressive) { + throw new Error('usage: node scripts/compare-live-benchmarks.mjs --atomic= --progressive='); +} + +const [atomic, progressive] = await Promise.all([ + readReport(args.atomic, 'atomic'), + readReport(args.progressive, 'progressive'), +]); +const comparison = compareModelBackedReports(atomic, progressive, { + medianTarget: ratioArg(args.medianTarget, 0.35), + p95Target: ratioArg(args.p95Target, 0.25), +}); + +process.stdout.write(JSON.stringify(comparison, null, 2) + '\n'); +if (!comparison.passed) process.exitCode = 1; + +async function readReport(file, delivery) { + const value = JSON.parse(await readFile(resolve(String(file)), 'utf-8')); + const reports = Array.isArray(value?.reports) ? value.reports : [value]; + const report = reports.find((item) => item?.benchmark?.delivery === delivery); + if (!report) throw new Error(`${file} does not contain a ${delivery} benchmark report`); + return report; +} + +function parseArgs(argv) { + const out = {}; + for (const arg of argv) { + if (!arg.startsWith('--')) continue; + const index = arg.indexOf('='); + if (index > 2) out[arg.slice(2, index)] = arg.slice(index + 1); + } + return out; +} + +function ratioArg(value, fallback) { + if (value == null) return fallback; + const parsed = Number(value); + if (!Number.isFinite(parsed) || parsed < 0 || parsed >= 1) throw new Error(`invalid threshold ratio: ${value}`); + return parsed; +} diff --git a/scripts/lib/live-benchmark.mjs b/scripts/lib/live-benchmark.mjs new file mode 100644 index 000000000..7f3b7911a --- /dev/null +++ b/scripts/lib/live-benchmark.mjs @@ -0,0 +1,395 @@ +import { performance } from 'node:perf_hooks'; + +const METRIC_KEYS = [ + 'browserPreparationMs', + 'browserDispatchMs', + 'automationClickMs', + 'serverPickupMs', + 'goToAgentMs', + 'serverPreflightMs', + 'scaffoldMs', + 'generationToFirstMs', + 'generationMs', + 'firstVariantWriteMs', + 'writeMs', + 'writeToFirstVariantMs', + 'replyMs', + 'goToFirstVariantMs', + 'goToAllVariantsMs', + 'deliveryGapMs', + 'impeccableOverheadMs', +]; + +export function createTraceRecorder(now = () => performance.now()) { + const events = []; + return { + events, + trace(name, data = {}) { + events.push({ name, at: now(), ...data }); + }, + mark(name, data = {}) { + const event = { name, at: now(), ...data }; + events.push(event); + return event; + }, + }; +} + +export function durationBetween(events, startName, endName, predicate = () => true) { + const start = events.find((event) => event.name === startName && predicate(event)); + const end = events.find((event) => event.name === endName && predicate(event) && (!start || event.at >= start.at)); + if (!start || !end) return null; + return roundMs(Math.max(0, end.at - start.at)); +} + +/** + * Assemble the two model calls used by the Live benchmark's progressive path. + * The first checkpoint is already visible in the browser, so both its markup + * and CSS are immutable. The tail call may supply deferred params for variant + * 1, but its CSS must contain only independently-scoped rules for variants 2+. + */ +export function assembleSplitProgressiveOutput(first, remaining) { + const firstVariant = first?.variants?.[0]; + if (!firstVariant) throw new Error('progressive assembly requires a first variant'); + if (!Array.isArray(remaining?.variants) || remaining.variants.length < 1) { + throw new Error('progressive assembly requires a complete remaining variant set'); + } + + const firstCss = String(first.scopedCss || ''); + const laterCss = String(remaining.scopedCss || ''); + assertLaterVariantCss(laterCss); + + return { + scopedCss: firstCss && laterCss ? `${firstCss}\n${laterCss}` : firstCss || laterCss, + variants: [ + { + ...firstVariant, + params: Array.isArray(remaining.variants[0]?.params) + ? remaining.variants[0].params + : [], + }, + ...remaining.variants.slice(1), + ], + }; +} + +export function buildInteractionRun(events, { iteration, scenario, goStartedAt, browserTiming = null }) { + const received = events.find((event) => + event.name === 'agent.event.received' + && event.type === 'generate' + && event.at >= goStartedAt + ); + if (!received?.id) throw new Error(`iteration ${iteration}: no generate event was traced`); + + const id = received.id; + const forId = (event) => event.id === id; + const eventPost = events.find((event) => event.name === 'browser.generate_post' && forId(event)); + const mark = (name) => events.find((event) => event.name === name && event.iteration === iteration); + const first = mark('browser.first_variant'); + const all = mark('browser.all_variants'); + const writeEnd = events.find((event) => event.name === 'agent.write.end' && forId(event)); + const firstWriteEnd = events.find((event) => event.name === 'agent.first_variant.write.end' && forId(event)); + const reusedScaffold = events.find((event) => event.name === 'agent.scaffold.reused' && forId(event)); + const generationMs = durationBetween(events, 'agent.generate.start', 'agent.generate.end', forId); + const generationToFirstMs = durationBetween(events, 'agent.generate.start', 'agent.generate.first_ready', forId); + const browserPreparationMs = eventPost ? roundMs(eventPost.at - goStartedAt) : null; + const browserDispatchMs = Number.isFinite(browserTiming?.goAt) && Number.isFinite(browserTiming?.generateAt) + ? roundMs(Math.max(0, browserTiming.generateAt - browserTiming.goAt)) + : null; + const interactionStartedAt = eventPost && browserDispatchMs != null + ? eventPost.at - browserDispatchMs + : goStartedAt; + const measuredGoToFirstVariantMs = first ? roundMs(first.at - interactionStartedAt) : null; + const measuredGoToAllVariantsMs = all ? roundMs(all.at - interactionStartedAt) : null; + + return { + iteration, + scenario, + eventId: id, + annotationEvidence: { + screenshotPath: eventPost?.hasScreenshotPath === true, + comments: Number(eventPost?.commentCount || 0), + strokes: Number(eventPost?.strokeCount || 0), + }, + browserPreparationMs, + browserDispatchMs, + automationClickMs: browserPreparationMs == null || browserDispatchMs == null + ? null + : roundMs(Math.max(0, browserPreparationMs - browserDispatchMs)), + serverPickupMs: eventPost ? roundMs(Math.max(0, received.at - eventPost.at)) : null, + goToAgentMs: roundMs(received.at - interactionStartedAt), + serverPreflightMs: Number.isFinite(reusedScaffold?.durationMs) ? roundMs(reusedScaffold.durationMs) : null, + scaffoldMs: durationBetween(events, 'agent.scaffold.start', 'agent.scaffold.end', forId), + generationToFirstMs, + generationMs, + firstVariantWriteMs: durationBetween(events, 'agent.first_variant.write.start', 'agent.first_variant.write.end', forId), + writeMs: durationBetween(events, 'agent.write.start', 'agent.write.end', forId), + writeToFirstVariantMs: first && (firstWriteEnd || writeEnd) + ? roundMs(Math.max(0, first.at - (firstWriteEnd || writeEnd).at)) + : null, + replyMs: durationBetween(events, 'agent.reply.start', 'agent.reply.end', forId), + goToFirstVariantMs: measuredGoToFirstVariantMs, + goToAllVariantsMs: measuredGoToAllVariantsMs, + deliveryGapMs: first && all ? roundMs(Math.max(0, all.at - first.at)) : null, + impeccableOverheadMs: measuredGoToFirstVariantMs == null || generationToFirstMs == null + ? null + : roundMs(Math.max(0, measuredGoToFirstVariantMs - generationToFirstMs)), + }; +} + +export function summarizeRuns(runs) { + const metrics = {}; + for (const key of METRIC_KEYS) { + const values = runs.map((run) => run[key]).filter(Number.isFinite).sort((a, b) => a - b); + if (values.length === 0) continue; + metrics[key] = { + median: roundMs(percentile(values, 0.5)), + p95: roundMs(percentile(values, 0.95)), + min: roundMs(values[0]), + max: roundMs(values[values.length - 1]), + }; + } + return { count: runs.length, metrics }; +} + +export function summarizeSetup(events) { + const stages = [ + ['dependencies', 'setup.install.start', 'setup.install.end'], + ['liveServer', 'setup.live_server.start', 'setup.live_server.end'], + ['injection', 'setup.inject.start', 'setup.inject.end'], + ['devServer', 'setup.dev_server.start', 'setup.dev_server.end'], + ['pageLoad', 'setup.page_load.start', 'setup.page_load.end'], + ['handshake', 'setup.handshake.start', 'setup.handshake.end'], + ]; + return Object.fromEntries(stages.map(([key, start, end]) => [key, durationBetween(events, start, end)])); +} + +export function createBenchmarkReport({ + fixture, + agent, + provider, + model, + scenario, + runs, + events, + harnessProbe = null, + delivery = 'atomic', + promptMode = null, + simulation = null, + generatedAt = new Date().toISOString(), +}) { + return { + schemaVersion: 1, + generatedAt, + benchmark: { + fixture, + agent, + provider: provider || null, + model: model || null, + scenario, + variants: 3, + delivery, + promptMode, + simulation, + }, + setup: summarizeSetup(events), + summary: summarizeRuns(runs), + runs, + harnessProbe, + }; +} + +export function mergeBenchmarkReports(reports, generatedAt = new Date().toISOString()) { + return { + schemaVersion: 1, + generatedAt, + reports, + }; +} + +export function compareModelBackedReports(atomic, progressive, { + medianTarget = 0.35, + p95Target = 0.25, + minimumRuns = 3, +} = {}) { + assertComparableModelReport(atomic, 'atomic', minimumRuns); + assertComparableModelReport(progressive, 'progressive', minimumRuns); + + for (const key of ['fixture', 'provider', 'model', 'scenario', 'variants', 'promptMode']) { + if (atomic.benchmark[key] !== progressive.benchmark[key]) { + throw new Error(`benchmark mismatch for ${key}: atomic=${atomic.benchmark[key]} progressive=${progressive.benchmark[key]}`); + } + } + + const atomicFirst = requiredMetric(atomic, 'goToFirstVariantMs'); + const progressiveFirst = requiredMetric(progressive, 'goToFirstVariantMs'); + const medianImprovement = improvement(atomicFirst.median, progressiveFirst.median); + const p95Improvement = improvement(atomicFirst.p95, progressiveFirst.p95); + const allReady = { + atomic: requiredMetric(atomic, 'goToAllVariantsMs'), + progressive: requiredMetric(progressive, 'goToAllVariantsMs'), + }; + const passed = medianImprovement >= medianTarget && p95Improvement >= p95Target; + + return { + passed, + target: { medianImprovement, p95Improvement, medianTarget, p95Target }, + firstReviewable: { atomic: atomicFirst, progressive: progressiveFirst }, + allVariantsReady: allReady, + benchmark: { + fixture: atomic.benchmark.fixture, + provider: atomic.benchmark.provider, + model: atomic.benchmark.model, + scenario: atomic.benchmark.scenario, + runs: { atomic: atomic.summary.count, progressive: progressive.summary.count }, + }, + }; +} + +function assertComparableModelReport(report, delivery, minimumRuns) { + if (!report?.benchmark || !report?.summary) throw new Error(`${delivery} benchmark report is missing metadata or summary`); + if (report.benchmark.agent !== 'llm') throw new Error(`${delivery} benchmark must be model-backed (agent=llm)`); + if (report.benchmark.delivery !== delivery) { + throw new Error(`expected ${delivery} delivery report, got ${report.benchmark.delivery || 'unknown'}`); + } + if (report.benchmark.simulation) throw new Error(`${delivery} model benchmark must not contain simulated latency`); + if (!report.benchmark.provider || !report.benchmark.model) throw new Error(`${delivery} benchmark is missing provider/model identity`); + if (!Number.isInteger(report.summary.count) || report.summary.count < minimumRuns) { + throw new Error(`${delivery} benchmark requires at least ${minimumRuns} runs`); + } +} + +function requiredMetric(report, key) { + const metric = report.summary.metrics?.[key]; + if (!Number.isFinite(metric?.median) || !Number.isFinite(metric?.p95)) { + throw new Error(`${report.benchmark.delivery} benchmark is missing ${key} median/p95`); + } + return { median: metric.median, p95: metric.p95 }; +} + +function improvement(baseline, candidate) { + if (!(baseline > 0) || !Number.isFinite(candidate)) throw new Error('benchmark latency must be finite and baseline must be positive'); + return Number((1 - (candidate / baseline)).toFixed(4)); +} + +function assertLaterVariantCss(css) { + if (!css.trim()) return; + for (const prelude of topLevelCssPreludes(css)) { + const variants = [...prelude.matchAll(/\[data-impeccable-variant\s*=\s*(["'])(\d+)\1[^\]]*\]/g)] + .map((match) => Number(match[2])); + if (variants.includes(1)) { + throw new Error('progressive tail CSS must not repeat or conflict with published variant 1 CSS'); + } + if (variants.length === 0 || variants.some((variant) => variant < 2)) { + throw new Error('progressive tail CSS must be attributable only to variants 2+'); + } + if (new Set(variants).size !== 1) { + throw new Error('each progressive tail CSS block must target exactly one later variant'); + } + } +} + +function topLevelCssPreludes(css) { + const preludes = []; + let cursor = 0; + while (cursor < css.length) { + while (cursor < css.length && /\s/.test(css[cursor])) cursor += 1; + if (cursor >= css.length) break; + const start = cursor; + const open = findCssToken(css, cursor, '{'); + if (open === -1) throw new Error('progressive tail CSS contains a rule without a block'); + const prelude = css.slice(start, open).trim(); + if (!prelude || prelude.includes(';')) { + throw new Error('progressive tail CSS must contain scoped rule blocks only'); + } + preludes.push(prelude); + const close = findMatchingCssBrace(css, open); + if (close === -1) throw new Error('progressive tail CSS has unbalanced braces'); + cursor = close + 1; + } + return preludes; +} + +function findCssToken(css, start, token) { + let quote = null; + let comment = false; + for (let index = start; index < css.length; index += 1) { + const char = css[index]; + const next = css[index + 1]; + if (comment) { + if (char === '*' && next === '/') { + comment = false; + index += 1; + } + continue; + } + if (!quote && char === '/' && next === '*') { + comment = true; + index += 1; + continue; + } + if (quote) { + if (char === '\\') index += 1; + else if (char === quote) quote = null; + continue; + } + if (char === '"' || char === "'") { + quote = char; + continue; + } + if (char === token) return index; + } + return -1; +} + +function findMatchingCssBrace(css, open) { + let depth = 0; + let quote = null; + let comment = false; + for (let index = open; index < css.length; index += 1) { + const char = css[index]; + const next = css[index + 1]; + if (comment) { + if (char === '*' && next === '/') { + comment = false; + index += 1; + } + continue; + } + if (!quote && char === '/' && next === '*') { + comment = true; + index += 1; + continue; + } + if (quote) { + if (char === '\\') index += 1; + else if (char === quote) quote = null; + continue; + } + if (char === '"' || char === "'") { + quote = char; + continue; + } + if (char === '{') depth += 1; + if (char === '}') { + depth -= 1; + if (depth === 0) return index; + } + } + return -1; +} + +function percentile(sortedValues, ratio) { + if (sortedValues.length === 1) return sortedValues[0]; + const index = (sortedValues.length - 1) * ratio; + const lower = Math.floor(index); + const upper = Math.ceil(index); + if (lower === upper) return sortedValues[lower]; + const weight = index - lower; + return sortedValues[lower] * (1 - weight) + sortedValues[upper] * weight; +} + +function roundMs(value) { + if (!Number.isFinite(value)) return null; + return Number(value.toFixed(2)); +} diff --git a/scripts/lib/live-provider-benchmark.mjs b/scripts/lib/live-provider-benchmark.mjs new file mode 100644 index 000000000..ac74b352d --- /dev/null +++ b/scripts/lib/live-provider-benchmark.mjs @@ -0,0 +1,547 @@ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { performance } from 'node:perf_hooks'; + +import { anthropic } from '@ai-sdk/anthropic'; +import { google } from '@ai-sdk/google'; +import { openai } from '@ai-sdk/openai'; +import { generateText } from 'ai'; + +import { + VARIANT_SYSTEM_INSTRUCTIONS, + parseVariantResponse, + validateProgressiveVariantOutput, + validateVariantCount, + validateVariantMaterialChange, + validateVariantVisibleCopy, +} from '../../tests/live-e2e/agents/llm-agent.mjs'; + +export const PROVIDER_PROFILES = Object.freeze({ + anthropic: { + label: 'Anthropic', + model: 'claude-sonnet-4-6', + envKeys: ['ANTHROPIC_API_KEY'], + pricePerMillion: { input: 3, cachedInput: 0.3, output: 15 }, + effort: 'low', + priceSource: 'https://platform.claude.com/docs/en/about-claude/pricing', + }, + openai: { + label: 'OpenAI', + model: 'gpt-5.5', + envKeys: ['OPENAI_API_KEY'], + pricePerMillion: { input: 5, cachedInput: 0.5, output: 30 }, + effort: 'low', + priceSource: 'https://developers.openai.com/api/docs/models/gpt-5.5', + }, + google: { + label: 'Google', + model: 'gemini-3.1-flash-lite', + envKeys: ['GOOGLE_GENERATIVE_AI_API_KEY', 'GOOGLE_CLOUD_API_KEY', 'GEMINI_API_KEY'], + pricePerMillion: { input: 0.25, cachedInput: 0.025, output: 1.5 }, + effort: 'minimal (provider default)', + priceSource: 'https://ai.google.dev/gemini-api/docs/pricing', + }, +}); + +export const STRATEGIES = Object.freeze({ + 'atomic-full': { + delivery: 'atomic', + promptMode: 'full-live-context', + calls: 'one 3-variant call', + }, + 'progressive-full': { + delivery: 'progressive', + promptMode: 'full-live-context', + calls: 'one first-variant call, then one remaining-directions call; deterministic assembly preserves variant 1', + }, + 'progressive-compact': { + delivery: 'progressive', + promptMode: 'compact-producer-contract', + calls: 'one first-variant call, then one remaining-directions call; deterministic assembly preserves variant 1', + }, + 'parallel-compact': { + delivery: 'parallel-progressive', + promptMode: 'compact-producer-contract', + calls: 'three concurrent one-variant calls; first valid result publishes immediately', + }, +}); + +export const BRAND_CONTRACT = Object.freeze({ + identity: 'Warm paper, dark ink, moss and brass accents; Georgia display with a restrained sans body; editorial, practical, and quiet.', + requiredCopy: [ + 'Quarterly print edition', + 'Field Notes', + 'Four routes, annotated maps, and practical details for unhurried weekends.', + 'Reserve issue eight', + ], + requiredClasses: [ + 'offer-card', + 'offer-card__copy', + 'offer-card__eyebrow', + 'offer-card__title', + 'offer-card__body', + 'action-link', + ], + allowedTokens: [ + '--color-paper', + '--color-paper-deep', + '--color-ink', + '--color-moss', + '--color-brass', + '--font-display', + '--font-body', + '--space-1', + '--space-2', + '--space-3', + '--space-4', + '--radius-control', + ], + sourceExcerpt: [ + '
', + '
', + '

Quarterly print edition

', + '

Field Notes

', + '

Four routes, annotated maps, and practical details for unhurried weekends.

', + '
', + ' Reserve issue eight', + '
', + ].join('\n'), +}); + +const COMPACT_CONTRACT = [ + VARIANT_SYSTEM_INSTRUCTIONS, + '', + 'QUALITY GATE FOR THIS PRODUCER:', + '- Preserve all visible copy exactly and retain the article/component class contract.', + '- Stay inside the supplied identity. Reuse the supplied CSS custom properties instead of inventing colors, typefaces, spacing, or radii.', + '- Do not add gradients, blur, glow, glass, neon, decorative shadows, emoji, or unrelated content.', + '- Make each variant materially different through hierarchy, layout, density, or color-role allocation.', +].join('\n'); + +export function loadBenchmarkEnv({ repoRoot, explicitPath } = {}) { + const candidates = [ + explicitPath, + repoRoot && path.join(repoRoot, '.env'), + path.join(os.homedir(), 'code', 'impeccable-evals', '.env'), + ].filter(Boolean); + const loaded = []; + for (const file of candidates) { + if (!fs.existsSync(file)) continue; + const body = fs.readFileSync(file, 'utf-8'); + for (const line of body.split(/\r?\n/)) { + const match = line.match(/^\s*([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(.*?)\s*$/); + if (!match || match[1].startsWith('#')) continue; + let value = match[2]; + if ((value.startsWith('"') && value.endsWith('"')) || (value.startsWith("'") && value.endsWith("'"))) { + value = value.slice(1, -1); + } + if (!process.env[match[1]] && value) process.env[match[1]] = value; + } + loaded.push(file); + } + if (!process.env.GOOGLE_GENERATIVE_AI_API_KEY) { + process.env.GOOGLE_GENERATIVE_AI_API_KEY = process.env.GOOGLE_CLOUD_API_KEY || process.env.GEMINI_API_KEY; + } + return loaded; +} + +export function resolveProviderSelection(providerNames, modelOverrides = {}) { + return providerNames.map((provider) => { + const profile = PROVIDER_PROFILES[provider]; + if (!profile) throw new Error(`unknown provider ${JSON.stringify(provider)}`); + const keyPresent = profile.envKeys.some((key) => Boolean(process.env[key])); + return { + provider, + label: profile.label, + model: modelOverrides[provider] || profile.model, + keyPresent, + pricePerMillion: profile.pricePerMillion, + effort: profile.effort, + priceSource: profile.priceSource, + }; + }); +} + +export function createProviderLiveAgent({ provider, model, strategy, liveSpec, onRecord = () => {} }) { + const strategyConfig = STRATEGIES[strategy]; + if (!strategyConfig) throw new Error(`unknown strategy ${JSON.stringify(strategy)}`); + const languageModel = providerModel(provider, model); + const system = strategyConfig.promptMode === 'full-live-context' + ? `${COMPACT_CONTRACT}\n\nFULL LIVE CONTEXT:\n${liveSpec}` + : COMPACT_CONTRACT; + const pendingParallel = new Map(); + const pendingFirst = new Map(); + + const request = async ({ event, phase, lane = null, firstVariant = null }) => { + const startedAt = performance.now(); + const expectedCount = Number(event.count); + const payload = benchmarkPayload(event, { phase, lane, firstVariant }); + const basePrompt = [ + 'Produce Impeccable Live variant output for this request. Return only the JSON object.', + phaseInstructions(phase, expectedCount, lane), + '', + '', + JSON.stringify(payload, null, 2), + '', + ].join('\n'); + let prompt = basePrompt; + let lastError; + for (let attempt = 1; attempt <= 2; attempt += 1) { + const attemptStartedAt = performance.now(); + let usage = null; + try { + const response = await generateText({ + model: languageModel, + system, + prompt, + maxOutputTokens: 12_000, + ...providerLatencyOptions(provider), + }); + usage = normalizeUsage(response.usage); + const parsed = parseVariantResponse(response.text); + const validationError = validateVariantOutput(parsed, event, { phase, firstVariant }); + if (validationError) throw new Error(validationError); + const record = { + provider, + model, + strategy, + phase, + lane, + attempt, + durationMs: roundMs(performance.now() - attemptStartedAt), + totalPhaseMs: roundMs(performance.now() - startedAt), + usage, + estimatedCostUsd: estimateCostUsd(usage, PROVIDER_PROFILES[provider].pricePerMillion), + output: parsed, + }; + onRecord(record); + return parsed; + } catch (error) { + lastError = error; + onRecord({ + provider, + model, + strategy, + phase, + lane, + attempt, + durationMs: roundMs(performance.now() - attemptStartedAt), + usage, + estimatedCostUsd: usage ? estimateCostUsd(usage, PROVIDER_PROFILES[provider].pricePerMillion) : 0, + error: String(error?.message || error), + }); + prompt = `${basePrompt}\n\nVALIDATION ERROR:\n${String(error?.message || error)}\nReturn corrected JSON only.`; + } + } + throw lastError; + }; + + if (strategy === 'atomic-full') { + return { + async generateVariants(event) { + return request({ event, phase: 'atomic' }); + }, + }; + } + + if (strategy === 'parallel-compact') { + return { + async generateFirstVariant(event) { + const lanes = ['hierarchy', 'layout', 'density']; + const calls = lanes.map((lane) => { + const laneEvent = { ...event, count: 1 }; + return request({ event: laneEvent, phase: 'parallel-lane', lane }).then((output) => ({ lane, output })); + }); + const first = await Promise.race(calls); + pendingParallel.set(event.id, { calls, first }); + return first.output; + }, + async generateRemainingVariants(event) { + const pending = pendingParallel.get(event.id); + if (!pending) throw new Error(`parallel generation state missing for ${event.id}`); + const settled = await Promise.all(pending.calls); + pendingParallel.delete(event.id); + const ordered = [pending.first, ...settled.filter((item) => item !== pending.first)]; + const variants = ordered.map((item) => item.output.variants[0]); + const scopedCss = ordered.map((item, index) => remapSingleVariantCss(item.output.scopedCss, index + 1)).join('\n'); + const output = { scopedCss, variants }; + onRecord({ provider, model, strategy, phase: 'parallel-assembled', lane: null, attempt: 1, usage: normalizeUsage(), estimatedCostUsd: 0, output }); + return output; + }, + }; + } + + return { + async generateFirstVariant(event) { + const first = await request({ event: { ...event, count: 1 }, phase: 'first' }); + pendingFirst.set(event.id, first); + return first; + }, + async generateRemainingVariants(event, context) { + const first = pendingFirst.get(event.id) || context.firstOutput; + if (!first?.variants?.[0]) throw new Error(`first variant state missing for ${event.id}`); + const remaining = await request({ + event: { ...event, count: Math.max(1, event.count - 1) }, + phase: 'remaining-directions', + firstVariant: first.variants[0], + }); + pendingFirst.delete(event.id); + return assembleProgressiveOutput(first, remaining); + }, + }; +} + +export function scoreVariantOutput(output, { validationError = null } = {}) { + const variants = Array.isArray(output?.variants) ? output.variants : []; + const css = String(output?.scopedCss || ''); + const perVariant = variants.map((variant) => String(variant.innerHtml || '')); + const copyChecks = perVariant.flatMap((html) => BRAND_CONTRACT.requiredCopy.map((copy) => html.includes(copy))); + const componentChecks = perVariant.flatMap((html) => [ + /^\s* new RegExp(`\\b${escapeRegExp(className)}\\b`).test(html)), + /href=["']#edition["']/.test(html), + /aria-labelledby=["']field-notes-title["']/.test(html), + ]); + const usedTokens = BRAND_CONTRACT.allowedTokens.filter((token) => css.includes(`var(${token}`)); + const rawColors = css.match(/#[0-9a-f]{3,8}\b|\b(?:rgb|hsl|oklch|lab)\s*\(/gi) || []; + const foreignFonts = css.match(/font-family\s*:\s*([^;}]+)/gi) || []; + const tokenChecks = [ + usedTokens.length >= 3, + rawColors.length === 0, + foreignFonts.every((declaration) => /var\(--font-(?:display|body)\)|inherit|serif|sans-serif/.test(declaration)), + !/\b(?:margin|padding|gap|border-radius)\s*:\s*(?!var\(|0(?:\D|$))[^;}]+/i.test(css), + ]; + const brandChecks = [ + /var\(--color-(?:paper|paper-deep|ink|moss|brass)\)/.test(css), + !/(?:linear|radial|conic)-gradient|backdrop-filter|filter\s*:\s*blur|text-shadow|box-shadow/i.test(css), + !/\b(?:neon|glass|glow|purple|magenta|cyan)\b/i.test(`${css}\n${perVariant.join('\n')}`), + !/border-radius\s*:\s*(?:999|[5-9]\d)px/i.test(css), + ]; + const sourceChecks = [ + !validationError, + variants.length > 0, + perVariant.every((html) => !/data-impeccable-| /^\s*\s*$/i.test(html)), + ]; + const dimensions = { + brandFidelity: dimension(brandChecks), + componentFidelity: dimension(componentChecks), + tokenFidelity: dimension(tokenChecks), + copyFidelity: dimension(copyChecks), + sourceValidity: dimension(sourceChecks), + }; + const overall = roundScore(Object.values(dimensions).reduce((sum, value) => sum + value, 0) / Object.keys(dimensions).length); + return { + ...dimensions, + overall, + passed: overall >= 0.9 && Object.values(dimensions).every((value) => value >= 0.75), + diagnostics: { + usedTokens, + rawColorCount: rawColors.length, + validationError, + }, + }; +} + +export function validateAcceptedCleanup({ source, browserClean, buildPassed, expectedCopy = BRAND_CONTRACT.requiredCopy }) { + const markerFree = !/data-impeccable-|impeccable-(?:variants|carbonize|params|original)/i.test(source); + const copyPreserved = expectedCopy.every((copy) => source.includes(copy)); + const sourceShape = /]*\boffer-card\b[\s\S]*<\/article>/.test(source); + const checks = { markerFree, copyPreserved, sourceShape, browserClean: Boolean(browserClean), buildPassed: Boolean(buildPassed) }; + return { ...checks, passed: Object.values(checks).every(Boolean) }; +} + +export function applyRuntimeSourceScore(quality, cleanup) { + const sourceChecks = [cleanup.markerFree, cleanup.copyPreserved, cleanup.sourceShape, cleanup.browserClean, cleanup.buildPassed]; + const sourceValidity = dimension(sourceChecks); + const dimensions = { + brandFidelity: quality.brandFidelity, + componentFidelity: quality.componentFidelity, + tokenFidelity: quality.tokenFidelity, + copyFidelity: quality.copyFidelity, + sourceValidity, + }; + const overall = roundScore(Object.values(dimensions).reduce((sum, value) => sum + value, 0) / Object.keys(dimensions).length); + return { + ...quality, + ...dimensions, + overall, + passed: cleanup.passed === true && overall >= 0.9 && Object.values(dimensions).every((value) => value >= 0.75), + }; +} + +export function assembleProgressiveOutput(first, remaining) { + if (!first?.variants?.[0]) throw new Error('progressive assembly requires a first variant'); + if (!Array.isArray(remaining?.variants) || remaining.variants.length === 0) { + throw new Error('progressive assembly requires remaining variants'); + } + return { + scopedCss: [first.scopedCss, shiftVariantCss(remaining.scopedCss, 1)].filter(Boolean).join('\n'), + variants: [first.variants[0], ...remaining.variants], + }; +} + +export function summarizeProviderRuns(runs) { + const latencyKeys = ['firstReviewableMs', 'allReadyMs', 'acceptCleanupMs']; + const metrics = {}; + for (const key of latencyKeys) { + const values = runs.map((run) => run[key]).filter(Number.isFinite).sort((a, b) => a - b); + if (values.length) metrics[key] = summarizeNumbers(values); + } + const qualityKeys = ['brandFidelity', 'componentFidelity', 'tokenFidelity', 'copyFidelity', 'sourceValidity', 'overall']; + const quality = {}; + for (const key of qualityKeys) { + const values = runs.map((run) => run.quality?.[key]).filter(Number.isFinite).sort((a, b) => a - b); + if (values.length) quality[key] = summarizeNumbers(values); + } + return { + count: runs.length, + metrics, + quality, + cleanupPassRate: runs.length ? roundScore(runs.filter((run) => run.cleanup?.passed).length / runs.length) : 0, + gatePassRate: runs.length ? roundScore(runs.filter((run) => run.passed).length / runs.length) : 0, + estimatedCostUsd: roundUsd(runs.reduce((sum, run) => sum + Number(run.estimatedCostUsd || 0), 0)), + }; +} + +function providerModel(provider, model) { + if (provider === 'anthropic') return anthropic(model); + if (provider === 'openai') return openai(model); + if (provider === 'google') return google(model); + throw new Error(`unsupported provider ${provider}`); +} + +function providerLatencyOptions(provider) { + if (provider === 'anthropic') return { providerOptions: { anthropic: { effort: 'low' } } }; + if (provider === 'openai') return { providerOptions: { openai: { reasoningEffort: 'low' } } }; + // Gemini 3.1 Flash-Lite defaults to minimal thinking; leaving the provider + // option unset preserves that latency-oriented default across SDK versions. + return {}; +} + +function benchmarkPayload(event, { phase, lane, firstVariant }) { + return { + request: { + id: event.id, + action: event.action, + freeformPrompt: event.freeformPrompt, + count: event.count, + phase, + lane, + firstVariant, + }, + pickedElement: event.element, + identityLock: BRAND_CONTRACT.identity, + sourceExcerpt: BRAND_CONTRACT.sourceExcerpt, + availableTokens: BRAND_CONTRACT.allowedTokens, + componentContract: { + rootTag: 'article', + requiredClasses: BRAND_CONTRACT.requiredClasses, + requiredHref: '#edition', + requiredAriaLabelledby: 'field-notes-title', + exactVisibleCopy: BRAND_CONTRACT.requiredCopy, + }, + }; +} + +function phaseInstructions(phase, count, lane) { + if (phase === 'first') { + return 'Return exactly one variant. Use params: [] so tunables stay off the first-reviewable path.'; + } + if (phase === 'remaining-directions') { + return `Return exactly ${count} new variants for different axes. Do not reproduce request.firstVariant; assembly preserves that first output byte-for-byte.`; + } + if (phase === 'parallel-lane') { + return `Return exactly one complete variant whose primary difference axis is ${lane}. It must stand alone and may include 0-3 useful params.`; + } + return `Return exactly ${count} complete variants in one response.`; +} + +function validateVariantOutput(parsed, event, { phase, firstVariant }) { + const phaseEvent = phase === 'first' + ? { ...event, progressive: { phase: 'first', totalCount: 3 } } + : event; + return validateVariantCount(parsed, phaseEvent) + || validateProgressiveVariantOutput(parsed, phaseEvent) + || validateVariantVisibleCopy(parsed, event.element) + || validateVariantMaterialChange(parsed, event.element); +} + +function remapSingleVariantCss(css, variantNumber) { + return String(css) + .replaceAll('[data-impeccable-variant="1"]', `[data-impeccable-variant="${variantNumber}"]`) + .replaceAll("[data-impeccable-variant='1']", `[data-impeccable-variant='${variantNumber}']`); +} + +function shiftVariantCss(css, amount) { + return String(css).replace(/(data-impeccable-variant=["'])(\d+)(["'])/g, (_, before, number, after) => { + return `${before}${Number(number) + amount}${after}`; + }); +} + +function normalizeUsage(usage = {}) { + const input = numberFrom(usage.inputTokens, usage.promptTokens, usage.inputTokenDetails?.noCacheTokens); + const cached = numberFrom(usage.cachedInputTokens, usage.inputTokenDetails?.cacheReadTokens, usage.inputTokenDetails?.cachedTokens); + const output = numberFrom(usage.outputTokens, usage.completionTokens); + return { + inputTokens: input, + cachedInputTokens: cached, + outputTokens: output, + totalTokens: numberFrom(usage.totalTokens, input + output), + }; +} + +export function estimateCostUsd(usage, pricing) { + const cached = Math.min(usage.cachedInputTokens || 0, usage.inputTokens || 0); + const uncached = Math.max(0, (usage.inputTokens || 0) - cached); + return roundUsd(( + uncached * pricing.input + + cached * pricing.cachedInput + + (usage.outputTokens || 0) * pricing.output + ) / 1_000_000); +} + +function numberFrom(...values) { + for (const value of values) if (Number.isFinite(value)) return Number(value); + return 0; +} + +function dimension(checks) { + return checks.length ? roundScore(checks.filter(Boolean).length / checks.length) : 0; +} + +function summarizeNumbers(values) { + return { + median: roundMs(percentile(values, 0.5)), + p95: roundMs(percentile(values, 0.95)), + min: roundMs(values[0]), + max: roundMs(values.at(-1)), + }; +} + +function percentile(values, ratio) { + if (values.length === 1) return values[0]; + const index = (values.length - 1) * ratio; + const lower = Math.floor(index); + const upper = Math.ceil(index); + if (lower === upper) return values[lower]; + return values[lower] + (values[upper] - values[lower]) * (index - lower); +} + +function escapeRegExp(value) { + return value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); +} + +function roundMs(value) { + return Number(Number(value).toFixed(2)); +} + +function roundScore(value) { + return Number(Number(value).toFixed(4)); +} + +function roundUsd(value) { + return Number(Number(value).toFixed(6)); +} diff --git a/site/data/live-annotated-result.ts b/site/data/live-annotated-result.ts new file mode 100644 index 000000000..390f5a9f1 --- /dev/null +++ b/site/data/live-annotated-result.ts @@ -0,0 +1,20 @@ +export const liveAnnotatedResult = { + generatedAt: '2026-07-13T00:16:44.377Z', + fixture: 'vite8-react-brand-fidelity', + provider: 'Anthropic', + model: 'Claude Haiku 4.5', + runs: 1, + delivery: 'progressive', + evidence: { + screenshot: true, + comments: 1, + strokes: 1, + }, + firstReviewableMs: 1264.75, + allVariantsMs: 7422.78, + impeccableOverheadMs: 119.94, + serverPickupMs: 49.69, + firstWriteMs: 7.72, + finalWriteMs: 5.95, + resetClean: true, +} as const; diff --git a/site/data/live-control-result.ts b/site/data/live-control-result.ts new file mode 100644 index 000000000..7327d832a --- /dev/null +++ b/site/data/live-control-result.ts @@ -0,0 +1,17 @@ +export const liveControlResult = { + generatedAt: '2026-07-12T03:17:04.109Z', + fixture: 'vite8-react-plain', + runs: 5, + acceptToPicking: { + medianMs: 1, + p95Ms: 2.8, + }, + nextGoToPickup: { + medianMs: 48, + p95Ms: 50.6, + }, + automation: { + acceptToPickingMedianMs: 266, + nextGoToPickupMedianMs: 864, + }, +}; diff --git a/site/data/live-framework-results.ts b/site/data/live-framework-results.ts new file mode 100644 index 000000000..62bf67602 --- /dev/null +++ b/site/data/live-framework-results.ts @@ -0,0 +1,29 @@ +export const liveFrameworkResults = [ + { + framework: 'Vite + React', + runs: 5, + acceptMedianMs: 1, + acceptP95Ms: 2.8, + nextPickupMedianMs: 48, + nextPickupP95Ms: 50.6, + status: 'Progressive E2E proven', + }, + { + framework: 'SvelteKit', + runs: 5, + acceptMedianMs: 2, + acceptP95Ms: 2, + nextPickupMedianMs: 315, + nextPickupP95Ms: 342, + status: 'Progressive E2E proven', + }, + { + framework: 'Nuxt 4 + Vue 3', + runs: 1, + acceptMedianMs: 143, + acceptP95Ms: 143, + nextPickupMedianMs: 116, + nextPickupP95Ms: 116, + status: 'Core + progressive E2E proven', + }, +]; diff --git a/site/data/live-harnesses.ts b/site/data/live-harnesses.ts index ac4a437d2..8b2a028d7 100644 --- a/site/data/live-harnesses.ts +++ b/site/data/live-harnesses.ts @@ -1,18 +1,18 @@ export const harnessPaths = [ { - name: 'Foreground long-poll', - harness: 'Codex desktop · observed', - pickup: 'Immediate when the turn is blocked', - tradeoff: 'Reliable, but Live occupies the main agent and inherits its model speed.', - status: 'Baseline', + name: 'Supervisor + generation worker', + harness: 'Codex desktop · canonical progressive path', + pickup: 'Concurrent control-event intake', + tradeoff: 'The foreground poll supervisor handles Accept/Discard and the next Go while a fenced worker generates variants.', + status: 'E2E proven', statusTone: 'ready', }, { - name: 'Background terminal + watcher', - harness: 'Codex app-server · available primitives', - pickup: 'Not yet measured end-to-end', - tradeoff: 'The app-server exposes output notifications and fs.watch, but a watcher still has to wake model work quickly.', - status: 'Benchmark next', + name: 'Foreground terminal supervisor', + harness: 'Codex desktop · unified exec session', + pickup: 'Concurrent shell work proven', + tradeoff: 'A yielded foreground process keeps running while other commands execute, but new output still needs an explicit session read; it does not wake a finished model turn by itself.', + status: 'Useful within active turn', statusTone: 'measure', source: 'https://learn.chatgpt.com/docs/app-server#api-overview', }, @@ -39,63 +39,103 @@ export const harnessPaths = [ export const liveExperiments = [ { rank: 1, + title: 'Accept any arrived variant — shipped', + evidence: 'A real Vite/React run accepts variant 1 while variants 2–3 are delayed, rejects the late worker write, returns to Pick, and leases a second Go under the 1.5 s gate.', + move: 'Durably fence generation on Accept/Discard, prioritize control events, publish through an epoch + source-hash transaction, and keep a separate poll supervisor active.', + expected: 'Removes the full-set wait and keeps the page interactive while canceled work unwinds.', + confidence: 'Protocol + browser E2E', + }, + { + rank: 2, + title: 'Remove the cold-start polling floor — shipped', + evidence: 'Ten configured Vite runs fell from 327.25 ms baseline to 153.03 ms cold median; p95 is 155.74 ms.', + move: 'Poll the detached helper readiness record every 5 ms instead of imposing a 200 ms minimum sleep.', + expected: 'Delivered a 53.2% cold-start reduction without changing injection or browser code.', + confidence: 'Measured + shipped', + }, + { + rank: 3, title: 'Dispatch first, capture second — shipped', evidence: 'Plain click-handler → generate fetch is now 2.2 ms median; end-to-end model-free latency fell from 916 ms to 414 ms.', move: 'Unannotated picks wait for the helper to accept the event, then capture the shader off-path. Annotated picks still capture and upload before dispatch.', expected: 'Delivered a 54.8% median reduction on the same fixture and deterministic agent.', confidence: 'Measured + shipped', }, - { - rank: 2, - title: 'Reveal variants progressively', - evidence: 'First and all variants arrive together today; the measured delivery gap is only the observer settling time.', - move: 'Give each variant its own ready event or preview file. Reveal variant 1 as soon as it validates while the remaining variants continue.', - expected: 'Large perceived win whenever generation is the dominant stage.', - confidence: 'Protocol change', - }, - { - rank: 3, - title: 'Keep a harness-native producer warm', - evidence: 'Codex and Claude can assign a different model to a custom subagent; both support continuing agent state.', - move: 'Resume one narrow Live producer with a compact contract, low reasoning, and source-write tools only. Quality-gate its output before preview.', - expected: 'Reduces model latency without forcing the user to change the main conversation model.', - confidence: 'Needs spawn/resume benchmark', - }, { rank: 4, - title: 'Move context work into the selection dwell', - evidence: 'The existing prefetch event was disabled because quick Go clicks paid an extra harness round trip.', - move: 'Debounce selection, then resolve the source file, style mode, token summary, and identity lock locally. Cancel or reuse the result on Go without queueing a model turn.', - expected: 'Shortens scaffold and prompt preparation while keeping quick clicks cheap.', - confidence: 'Low-risk prototype', + title: 'Reveal variants progressively — shipped', + evidence: 'With a simulated 2.00 s variants 2–3 tail, median first-reviewable latency fell from 2.13 s to 145 ms while full-set completion stayed flat.', + move: 'Codex publishes a monotonically growing prefix as each variant validates. The browser reveals arrived variants and pending dots immediately; Accept and Discard fence unfinished work.', + expected: 'Delivered a 93.2% median and 92.2% p95 perceived-latency reduction in the matched deterministic run.', + confidence: 'Measured + shipped', }, { rank: 5, - title: 'Generate knobs after pixels', - evidence: 'Parameter manifests are authored in the same response as every variant even though the first job is visual comparison.', - move: 'Render validated variants first. Infer coarse CSS knobs locally or ask the producer for parameters only for the visible/selected variant.', - expected: 'Cuts output tokens and lets the first useful preview land sooner.', - confidence: 'Quality experiment', + title: 'Preflight source locally — shipped', + evidence: 'The helper resolves and wraps the source in roughly 54 ms median as the Generate event is leased, before the model-facing event returns.', + move: 'Attach durable scaffold metadata to the queued event. Every harness reuses it; discovery failures preserve the existing agent-driven fallback.', + expected: 'Removes one deterministic source-discovery tool round trip from every successful generation.', + confidence: 'Measured + shipped', }, { rank: 6, - title: 'Parallelize variant ideas, centralize writes', - evidence: 'Modern harnesses can run several specialized agents concurrently, but shared-file edits create coordination risk.', - move: 'Have workers return one structured variant each; one coordinator validates and performs the only source write. Stream the first valid result.', - expected: 'Lower time to first option and stronger diversity at higher token cost.', - confidence: 'Expensive experiment', + title: 'Generate knobs after pixels — shipped', + evidence: 'The first progressive write intentionally carries no parameter manifest; tune controls arrive with the complete variant set.', + move: 'Plan parameter axes with the trio, but defer their manifests and CSS branches until the final delivery edit.', + expected: 'Keeps parameter output off the first-reviewable critical path without shrinking the final tuning surface.', + confidence: 'Protocol shipped', }, { rank: 7, - title: 'Trim and cache the producer contract', - evidence: 'The full Live reference is large and stable; only a small slice is needed for one generate event.', - move: 'Compile a provider-specific generation contract, keep the stable prefix cacheable, and attach only the action reference plus picked-element context.', - expected: 'Lower prefill latency and cost, especially for fresh subagents.', - confidence: 'Harness-specific', + title: 'Acknowledge first, clean up off-path — shipped', + evidence: 'React and Svelte release the picker in 1–2 ms after durable Accept. The provider-independent cleanup control reaches marker-free, buildable source in 171 ms with zero console errors.', + move: 'Keep the durable accept acknowledgement on the foreground path, then hand carbonize cleanup to a source-locked worker while the poll supervisor leases new work.', + expected: 'Preserves immediate interaction without leaving temporary Live source behind.', + confidence: 'Browser + build control', }, { rank: 8, - title: 'Wake on the journal, not terminal stdout', + title: 'Do not release before durability — rejected', + evidence: 'The measured durable acknowledgement already releases the picker in 1–2 ms. Releasing before it lands cannot create a perceptible win, but can show a successful Accept that recovery cannot replay.', + move: 'Keep the acknowledgement barrier; make every later cleanup and validation step asynchronous instead.', + expected: 'Retains crash recovery for no measurable interaction penalty.', + confidence: 'Measured floor + recovery gate', + }, + { + rank: 9, + title: 'Do not batch a whole Live session — rejected as default', + evidence: 'Immediate transactional Accept already returns control in 1–2 ms. Deferring source commits would move rather than remove work, make the next generation read stale design context, and enlarge the crash-recovery boundary.', + move: 'Batch only cleanup operations that do not affect the next generation. Commit each chosen design under the source lock before treating it as project truth.', + expected: 'Keeps later variants on-brand with the latest accepted source and bounds recovery to one interaction.', + confidence: 'Quality + recovery rejection', + }, + { + rank: 10, + title: 'Keep a harness-native producer warm — not the default', + evidence: 'A paired Codex probe produced a correct identity lock and guardrails, but resume took 23.1 s versus 21.7 s fresh. One pair is not statistically stable, and it shows no speed signal.', + move: 'Keep warm resume as an opt-in harness experiment; do not put it on the critical path until repeated spawn/resume measurements beat compact progressive generation.', + expected: 'Avoids paying context-management overhead for an unproven latency gain.', + confidence: 'Paired harness probe', + }, + { + rank: 11, + title: 'Parallelize ideas, centralize writes — optional fast path', + evidence: 'Strict-gated provider runs made parallel compact fastest for GPT-5.5 and Gemini 3.1 Flash-Lite. Every worker remained write-free; deterministic assembly performed the only publication.', + move: 'Use parallel compact only where extra calls are acceptable. Keep progressive compact as the portable default and preserve variant 1 byte-for-byte.', + expected: 'Lowers time to first option without shared-file races or brand drift.', + confidence: 'Paid provider smoke matrix', + }, + { + rank: 12, + title: 'Trim the producer contract — selected', + evidence: 'Progressive compact passed every measured brand, component, token, copy, source, and cleanup gate while cutting first-review latency 42–62% for Claude and GPT.', + move: 'Keep the stable generation contract cacheable and send only the picked-element context plus the current action and identity lock.', + expected: 'Reduces prefill and output work without weakening the design brief.', + confidence: 'Cross-provider quality gate', + }, + { + rank: 13, + title: 'Wake on the journal, not terminal stdout — still architectural', evidence: 'Codex app-server exposes fs.watch and process output notifications, while this desktop probe still required an explicit terminal read.', move: 'Benchmark a plugin/app-server bridge that watches the durable Live journal and starts or steers a dedicated turn directly.', expected: 'Frees the main turn if watcher-to-turn startup beats foreground polling.', @@ -106,8 +146,12 @@ export const liveExperiments = [ export const currentHarnessProbe = { testedAt: '2026-07-11', surface: 'Codex desktop unified exec', - delayedOutputMs: 3000, - processExitMs: 4000, + delayedOutputMs: 250, + processExitMs: null, + parallelCommandMs: 0.01, + parallelWorkSucceeded: true, + backgroundChildSurvivedShellExit: false, + requiresExplicitSessionRead: true, surfacedAutomatically: false, - result: 'The sentinel appeared only after an explicit session read; no app terminal was attached.', + result: 'A yielded foreground poller kept producing output while a separate command completed immediately. A traditional shell-backgrounded child did not survive shell exit. Output still required an explicit session read and did not proactively wake a finished turn.', }; diff --git a/site/data/live-init-result.ts b/site/data/live-init-result.ts new file mode 100644 index 000000000..e8d6f893b --- /dev/null +++ b/site/data/live-init-result.ts @@ -0,0 +1,14 @@ +export const liveInitResult = { + generatedAt: '2026-07-11T19:39:09-07:00', + fixture: 'vite8-react-plain', + runs: 10, + baselineColdMedianMs: 327.25, + cold: { + medianMs: 153.03, + p95Ms: 155.74, + }, + warm: { + medianMs: 72.8, + p95Ms: 75.4, + }, +}; diff --git a/site/data/live-progressive-result.ts b/site/data/live-progressive-result.ts new file mode 100644 index 000000000..e81096624 --- /dev/null +++ b/site/data/live-progressive-result.ts @@ -0,0 +1,25 @@ +export const progressiveDeliveryResult = { + generatedAt: '2026-07-11T18:41:20-07:00', + runs: 5, + benchmark: { + agent: 'llm', + provider: 'Anthropic', + model: 'Claude Haiku 4.5', + fixture: 'vite8-react-plain', + promptMode: 'synthetic-element-contract', + }, + atomic: { + medianFirstMs: 3674.74, + p95FirstMs: 7262.77, + medianAllMs: 3677.16, + firstMs: [3674.74, 3324.98, 3431.78, 4041.59, 8068.07], + allMs: [3677.16, 3327.62, 3433.45, 4045.33, 8070.09], + }, + progressive: { + medianFirstMs: 1306.33, + p95FirstMs: 1594.93, + medianAllMs: 4445.29, + firstMs: [1642, 1205.54, 1406.66, 1306.33, 1112.19], + allMs: [4770.53, 3127.94, 5037.59, 4031.08, 4445.29], + }, +}; diff --git a/site/data/live-provider-result.ts b/site/data/live-provider-result.ts new file mode 100644 index 000000000..a1207c752 --- /dev/null +++ b/site/data/live-provider-result.ts @@ -0,0 +1,78 @@ +export const liveProviderResult = { + generatedAt: '2026-07-12T03:22:46.816Z', + fixture: 'vite8-react-brand-fidelity', + runsPerCandidate: 1, + statisticallyStable: false, + paidSmokeMatrix: { + ran: ['atomic-full', 'progressive-compact', 'parallel-compact'], + notRun: [ + { + strategy: 'progressive-full', + reason: 'External execution credits were exhausted before the direct full-context split comparison could run. No result is inferred.', + }, + ], + }, + cleanupControl: { + providerIndependent: true, + acceptToCleanPickingMs: 171.41, + productionBuildMs: 378.04, + markerFree: true, + browserClean: true, + consoleErrors: 0, + passed: true, + }, + providers: [ + { + provider: 'Anthropic', + model: 'Claude Sonnet 4.6', + effort: 'Provider default in this sample; future harness runs explicitly use low.', + atomic: { firstMs: 32208.62, allMs: 32208.62, costUsd: 0.103779, quality: 0.9, passed: true }, + progressiveCompact: { firstMs: 12247.31, allMs: 31709.49, costUsd: 0.057795, quality: 0.9, passed: true }, + parallelCompact: { firstMs: 12361.28, allMs: 31378.19, costUsd: 0.064641, quality: 0.9, passed: true }, + progressiveFirstImprovement: 0.6198, + parallelFirstImprovement: 0.6162, + }, + { + provider: 'OpenAI', + model: 'GPT-5.5', + effort: 'low', + atomic: { firstMs: 22750.85, allMs: 22750.85, costUsd: 0.16728, quality: 0.95, passed: true }, + progressiveCompact: { firstMs: 13124.09, allMs: 24735.95, costUsd: 0.098515, quality: 0.9, passed: true }, + parallelCompact: { firstMs: 11846.73, allMs: 15871.87, costUsd: 0.145085, quality: 0.9, passed: true }, + progressiveFirstImprovement: 0.4231, + parallelFirstImprovement: 0.4793, + }, + { + provider: 'Google', + model: 'Gemini 3.1 Flash-Lite', + effort: 'minimal (provider default)', + atomic: { + firstMs: null, + allMs: null, + costUsd: null, + quality: null, + passed: false, + reason: 'Both attempts failed strict output validation.', + }, + progressiveCompact: { firstMs: 2654.97, allMs: 6764.95, costUsd: 0.002813, quality: 0.95, passed: true }, + parallelCompact: { firstMs: 1607.89, allMs: 1988.32, costUsd: 0.003956, quality: 0.95, passed: true }, + progressiveFirstImprovement: null, + parallelFirstImprovement: null, + }, + ], + cost: { + apiCalls: 22, + measuredLowerBoundUsd: 0.643864, + note: 'Five rejected validation responses predate failed-response usage capture and are excluded. Current runs capture their usage.', + }, + gate: { + dimensions: ['brand', 'component', 'token', 'copy', 'source', 'accept-cleanup'], + minimumOverall: 0.9, + minimumDimension: 0.75, + }, + recommendation: { + default: 'progressive-compact', + optionalFastPath: 'parallel-compact', + rationale: 'Progressive compact cut first-review latency 42–62% for Claude and GPT while preserving every quality gate. Parallel compact was strongest for GPT and Gemini but spends more requests.', + }, +} as const; diff --git a/site/pages/live-lab/index.astro b/site/pages/live-lab/index.astro index e4e121147..983979835 100644 --- a/site/pages/live-lab/index.astro +++ b/site/pages/live-lab/index.astro @@ -2,6 +2,12 @@ import Base from '../../layouts/Base.astro'; import benchmarkData from '../../data/live-performance.json'; import { currentHarnessProbe, harnessPaths, liveExperiments } from '../../data/live-harnesses'; +import { progressiveDeliveryResult } from '../../data/live-progressive-result'; +import { liveInitResult } from '../../data/live-init-result'; +import { liveControlResult } from '../../data/live-control-result'; +import { liveFrameworkResults } from '../../data/live-framework-results'; +import { liveProviderResult } from '../../data/live-provider-result'; +import { liveAnnotatedResult } from '../../data/live-annotated-result'; import '../../styles/sub-pages.css'; import '../../styles/live-performance.css'; @@ -16,10 +22,12 @@ const protocolFloor = metric('goToFirstVariantMs'); const baselineFloor = baseline.summary.metrics.goToFirstVariantMs?.median || protocolFloor; const browserDispatch = metric('browserDispatchMs'); const automationClick = metric('automationClickMs'); +const baselineAutomationClick = baseline.summary.metrics.automationClickMs?.median ?? automationClick; const serverPickup = metric('serverPickupMs'); const scaffold = metric('scaffoldMs'); const writeAndRender = metric('writeMs') + metric('writeToFirstVariantMs'); const productFloor = Math.max(0, protocolFloor - automationClick); +const baselineProductFloor = Math.max(0, baselineFloor - baselineAutomationClick); const improvement = baselineFloor ? Math.max(0, 1 - (protocolFloor / baselineFloor)) : 0; const annotatedFloor = annotated?.summary.metrics.goToFirstVariantMs?.median || 0; const maxScenarioFloor = Math.max(baselineFloor, protocolFloor, annotatedFloor, 1); @@ -40,6 +48,19 @@ const stageData = [ { name: 'Source scaffold', value: scaffold, tone: 'scaffold' }, { name: 'Write + browser settle', value: writeAndRender, tone: 'render' }, ]; +const progressiveGain = 1 - ( + progressiveDeliveryResult.progressive.medianFirstMs + / progressiveDeliveryResult.atomic.medianFirstMs +); +const progressiveP95Gain = 1 - ( + progressiveDeliveryResult.progressive.p95FirstMs + / progressiveDeliveryResult.atomic.p95FirstMs +); +const progressiveAllDelta = ( + progressiveDeliveryResult.progressive.medianAllMs + - progressiveDeliveryResult.atomic.medianAllMs +); +const initGain = 1 - (liveInitResult.cold.medianMs / liveInitResult.baselineColdMedianMs); ---
@@ -121,10 +142,144 @@ const stageData = [ ))} -
+
Result 01

Unannotated requests wait only for the helper to accept the event. Shader capture continues off-path; annotated requests still capture and upload first because the screenshot is semantic input.

+ +
+ + + + + {liveFrameworkResults.map(result => ( + + + + + + + ))} + +
Progressive Live performance by framework
FrameworkAccept → PickNext Go → pickupProof
{result.framework}{result.runs} real {result.runs === 1 ? 'run' : 'runs'}{displayMs(result.acceptMedianMs)}p95 {displayMs(result.acceptP95Ms)}{displayMs(result.nextPickupMedianMs)}p95 {displayMs(result.nextPickupP95Ms)}{result.status}
+
+ + +
+
+
+

Progressive delivery · Codex path

+

Review starts before generation ends

+
+

Variant 1 becomes reviewable and acceptable immediately. Accept fences the old worker, releases the picker, and lets the poll supervisor lease the next Go while later variants unwind.

+
+ +
+ Result 02 +

+ Median first-reviewable latency fell {(progressiveGain * 100).toFixed(1)}%; p95 fell {(progressiveP95Gain * 100).toFixed(1)}%. + Full-set completion changed by {progressiveAllDelta.toFixed(1)} ms. This is a five-run model-backed comparison using {progressiveDeliveryResult.benchmark.provider} {progressiveDeliveryResult.benchmark.model} and a sanitized synthetic heading contract. +

+
+
+ Control lane +

+ Across {liveControlResult.runs} real Vite/React runs, Accept dispatch → picker-ready is {displayMs(liveControlResult.acceptToPicking.medianMs)} median / {displayMs(liveControlResult.acceptToPicking.p95Ms)} p95. + A second Go reaches the poll supervisor in {displayMs(liveControlResult.nextGoToPickup.medianMs)} median / {displayMs(liveControlResult.nextGoToPickup.p95Ms)} p95 while the canceled worker unwinds. +

+
+
+ +
+
+
+

Paid provider smoke matrix

+

Decompose the work, preserve the first answer

+
+

One strict-gated run per candidate. Brand, component, token, copy, source, and cleanup fidelity must all pass.

+
+ +
+ + + + + {liveProviderResult.providers.map(result => ( + + + + + + + ))} + +
Paid Live generation strategy results by provider
ProviderAtomic fullProgressive compactParallel compact
{result.provider}{result.model} · {result.effort} + {result.atomic.firstMs === null ? 'Rejected' : displayMs(result.atomic.firstMs)} + {result.atomic.passed ? 'quality gate passed' : ('reason' in result.atomic ? result.atomic.reason : 'quality gate failed')} + + {displayMs(result.progressiveCompact.firstMs)} first + {displayMs(result.progressiveCompact.allMs)} all · gate passed + + {displayMs(result.parallelCompact.firstMs)} first + {displayMs(result.parallelCompact.allMs)} all · gate passed +
+
+
+ Decision +

+ Use progressive compact as the safe default: it cut first-review latency 42–62% for Claude and GPT without weakening any gate. Parallel compact is an optional GPT/Gemini fast path when extra calls and cost are acceptable. Variant 1 and its CSS are carried byte-for-byte into deterministic local assembly; later calls may only add directions. +

+
+

+ Provider matrix cost: ${liveProviderResult.cost.measuredLowerBoundUsd.toFixed(3)} lower bound. {liveProviderResult.cost.note}{' '} + The full-context progressive split was not run after external execution credits were exhausted; no result is inferred. Accept → clean Pick/build control passed in {displayMs(liveProviderResult.cleanupControl.acceptToCleanPickingMs)} with zero console errors or Live markers. +

+
+ +
+
+
+

Configured cold initialization

+

The helper stopped sleeping

+
+

Ten isolated starts. App dependency installation and dev-server startup are excluded.

+
+ +
+ Result 03 +

Cold median fell {(initGain * 100).toFixed(1)}% to {displayMs(liveInitResult.cold.medianMs)}; p95 is {displayMs(liveInitResult.cold.p95Ms)}. Warm median is {displayMs(liveInitResult.warm.medianMs)}.

+
@@ -153,6 +308,13 @@ const stageData = [

Median protocol measurements: {baseline.summary.count} before runs, {plain.summary.count} optimized runs, {annotated?.summary.count || 0} annotated controls. Browser and filesystem caches were warm.

+
+ Annotated model proof +

+ {liveAnnotatedResult.provider} {liveAnnotatedResult.model} received a real screenshot, {liveAnnotatedResult.evidence.comments} comment, and {liveAnnotatedResult.evidence.strokes} stroke. + First review arrived in {displayMs(liveAnnotatedResult.firstReviewableMs)} with {displayMs(liveAnnotatedResult.impeccableOverheadMs)} of measured Impeccable overhead; all variants arrived in {displayMs(liveAnnotatedResult.allVariantsMs)} and teardown returned to clean source. +

+
@@ -170,12 +332,12 @@ const stageData = [
Before optimization 15.9 s - model + measured baseline + model + measured floor, normalized by matched actionability control
Current optimized path 15.1 s - model + measured floor, excluding Playwright actionability + model + measured floor, normalized by matched actionability control
@@ -189,8 +351,10 @@ const stageData = [

Measure pickup before choosing architecture. Fast models cannot help while no model is running.

-
+ +
+ {harnessPaths.map(path => ( @@ -213,10 +377,10 @@ const stageData = [
-

Evidence-ranked backlog

-

What to test next

+

Evidence-ranked decisions

+

What survived testing

-

Actual latency first, perceived latency second, architectural bets last.

+

Shipped paths, rejected compromises, and unproven architectural bets are labeled separately.

    {liveExperiments.map(experiment => ( @@ -240,6 +404,7 @@ const stageData = [

    bun run bench:live boots a real framework fixture and Chromium, drives Pick → Go → Cycle, and records browser preparation, server pickup, scaffold, generation, write, and render boundaries. + node scripts/benchmark-live-init.mjs --iterations 10 measures configured cold and warm helper initialization separately. The deterministic agent makes Impeccable overhead visible; model-backed runs remain opt-in because they send fixture context to an external provider.

diff --git a/site/styles/live-performance.css b/site/styles/live-performance.css index 83a561052..43d09a15b 100644 --- a/site/styles/live-performance.css +++ b/site/styles/live-performance.css @@ -3,6 +3,25 @@ color: var(--ks-text); } +html.light .live-performance-page { + --ks-kinpaku: var(--ks-kinpaku-deep); + --ks-patina: var(--ks-patina-deep); + --ks-code-cmd: var(--ks-link-on-paper); + --ks-text-faint: var(--ks-text-muted); +} + +.live-performance-page .sr-only { + position: absolute; + width: 1px; + height: 1px; + padding: 0; + margin: -1px; + overflow: hidden; + clip: rect(0, 0, 0, 0); + white-space: nowrap; + border: 0; +} + .live-performance-main { overflow: clip; } @@ -19,6 +38,9 @@ } .live-lab-brand { + display: inline-flex; + align-items: center; + min-height: 44px; color: var(--ks-kinpaku); text-decoration: none; } @@ -254,6 +276,15 @@ border-bottom: 1px solid var(--ks-kinpaku-deep); } +.live-finding-card--quiet { + grid-template-columns: 112px minmax(0, 1fr); + gap: 22px; + padding: 22px 24px; + border: 1px solid var(--ks-rule); + border-radius: 6px; + background: var(--ks-lacquer-raised); +} + .live-performance-finding strong { color: var(--ks-code-cmd); font-family: var(--ks-mono); @@ -401,6 +432,15 @@ overflow-x: auto; } +.harness-table-wrap:focus-visible { + outline: 2px solid var(--ks-patina); + outline-offset: 4px; +} + +.table-scroll-hint { + display: none; +} + .harness-table { width: 100%; min-width: 840px; @@ -591,6 +631,58 @@ } @media (max-width: 560px) { + .live-performance-page { + --ks-type-eyebrow-size: var(--ks-type-body-size); + --ks-type-mono-size: var(--ks-type-body-size); + } + + .live-performance-meta { + font-size: var(--ks-type-body-size); + } + + .live-performance-page .ks-tag, + .live-performance-method code { + font-size: var(--ks-type-body-size); + } + + .table-scroll-hint { + display: block; + margin: 44px 0 -32px; + color: var(--ks-text-muted); + font-family: var(--ks-mono); + font-size: var(--ks-type-body-size); + } + + .harness-table-wrap { + margin-top: 44px; + padding-bottom: 8px; + scrollbar-color: var(--ks-patina) var(--ks-graphite); + } + + .harness-table-wrap::after { + position: sticky; + right: 0; + display: block; + width: 36px; + height: 3px; + margin-top: -3px; + margin-left: auto; + background: var(--ks-patina); + content: ''; + } + + .harness-table a { + display: inline-flex; + align-items: center; + min-height: 44px; + padding-inline: 3px; + } + + .live-simulator input[type='range'] { + min-height: 44px; + margin-top: 16px; + } + .live-performance-hero h1 { font-size: var(--ks-type-display-size); } diff --git a/tests/framework-fixtures/vite8-react-brand-fidelity/files/index.html b/tests/framework-fixtures/vite8-react-brand-fidelity/files/index.html new file mode 100644 index 000000000..797551452 --- /dev/null +++ b/tests/framework-fixtures/vite8-react-brand-fidelity/files/index.html @@ -0,0 +1,12 @@ + + + + + + Northstar Field Journal + + +
+ + + diff --git a/tests/framework-fixtures/vite8-react-brand-fidelity/files/package.json b/tests/framework-fixtures/vite8-react-brand-fidelity/files/package.json new file mode 100644 index 000000000..1cb4ace9d --- /dev/null +++ b/tests/framework-fixtures/vite8-react-brand-fidelity/files/package.json @@ -0,0 +1,18 @@ +{ + "name": "vite8-react-brand-fidelity-fixture", + "private": true, + "version": "0.0.0", + "type": "module", + "scripts": { + "dev": "vite --host 127.0.0.1", + "build": "vite build" + }, + "dependencies": { + "react": "^19.0.0", + "react-dom": "^19.0.0" + }, + "devDependencies": { + "@vitejs/plugin-react": "^6.0.0", + "vite": "^8.0.0" + } +} diff --git a/tests/framework-fixtures/vite8-react-brand-fidelity/files/src/App.jsx b/tests/framework-fixtures/vite8-react-brand-fidelity/files/src/App.jsx new file mode 100644 index 000000000..703910b80 --- /dev/null +++ b/tests/framework-fixtures/vite8-react-brand-fidelity/files/src/App.jsx @@ -0,0 +1,26 @@ +function ActionLink({ children }) { + return {children}; +} + +export default function App() { + return ( +
+
+

Northstar Field Journal

+

Useful observations from the long way around.

+
+ +
+

Edition 08 · Coastal paths

+
+
+

Quarterly print edition

+

Field Notes

+

Four routes, annotated maps, and practical details for unhurried weekends.

+
+ Reserve issue eight +
+
+
+ ); +} diff --git a/tests/framework-fixtures/vite8-react-brand-fidelity/files/src/main.jsx b/tests/framework-fixtures/vite8-react-brand-fidelity/files/src/main.jsx new file mode 100644 index 000000000..f2baba283 --- /dev/null +++ b/tests/framework-fixtures/vite8-react-brand-fidelity/files/src/main.jsx @@ -0,0 +1,10 @@ +import { StrictMode } from 'react'; +import { createRoot } from 'react-dom/client'; +import App from './App.jsx'; +import './styles.css'; + +createRoot(document.getElementById('root')).render( + + + , +); diff --git a/tests/framework-fixtures/vite8-react-brand-fidelity/files/src/styles.css b/tests/framework-fixtures/vite8-react-brand-fidelity/files/src/styles.css new file mode 100644 index 000000000..e4887dceb --- /dev/null +++ b/tests/framework-fixtures/vite8-react-brand-fidelity/files/src/styles.css @@ -0,0 +1,111 @@ +:root { + --color-paper: #f3efe4; + --color-paper-deep: #e7dfcf; + --color-ink: #20251f; + --color-moss: #526248; + --color-brass: #9b6b2f; + --font-display: Georgia, "Times New Roman", serif; + --font-body: Inter, Arial, sans-serif; + --space-1: 0.5rem; + --space-2: 1rem; + --space-3: 1.5rem; + --space-4: 2.5rem; + --radius-control: 0.25rem; +} + +* { box-sizing: border-box; } + +body { + margin: 0; + background: var(--color-paper); + color: var(--color-ink); + font-family: var(--font-body); +} + +.page-shell { + width: min(70rem, calc(100% - 2rem)); + margin: 0 auto; + padding: 5rem 0; +} + +.masthead { + max-width: 50rem; + margin-bottom: 4rem; +} + +.masthead__kicker, +.edition__number, +.offer-card__eyebrow { + color: var(--color-moss); + font-size: 0.75rem; + font-weight: 700; + letter-spacing: 0.12em; + text-transform: uppercase; +} + +h1, +h2 { + font-family: var(--font-display); + font-weight: 400; + text-wrap: balance; +} + +h1 { + margin: var(--space-2) 0 0; + font-size: clamp(3rem, 7vw, 5.5rem); + line-height: 0.98; +} + +.edition { + border-top: 1px solid var(--color-brass); + padding-top: var(--space-2); +} + +.offer-card { + display: grid; + grid-template-columns: minmax(0, 1fr) auto; + gap: var(--space-4); + align-items: end; + margin-top: var(--space-2); + padding: var(--space-4); + background: var(--color-paper-deep); + border-left: 0.25rem solid var(--color-moss); +} + +.offer-card__eyebrow, +.offer-card__body { + margin: 0; +} + +.offer-card__title { + margin: var(--space-1) 0; + font-size: 2.5rem; + line-height: 1; +} + +.offer-card__body { + max-width: 58ch; + line-height: 1.65; +} + +.action-link { + display: inline-flex; + min-height: 2.75rem; + align-items: center; + padding: 0 var(--space-3); + border: 1px solid var(--color-ink); + border-radius: var(--radius-control); + color: var(--color-ink); + font-weight: 700; + text-decoration: none; +} + +.action-link:focus-visible { + outline: 0.2rem solid var(--color-brass); + outline-offset: 0.2rem; +} + +@media (max-width: 42rem) { + .offer-card { grid-template-columns: 1fr; } + .action-link { justify-content: center; } +} diff --git a/tests/framework-fixtures/vite8-react-brand-fidelity/files/vite.config.js b/tests/framework-fixtures/vite8-react-brand-fidelity/files/vite.config.js new file mode 100644 index 000000000..dd5cfa6f0 --- /dev/null +++ b/tests/framework-fixtures/vite8-react-brand-fidelity/files/vite.config.js @@ -0,0 +1,10 @@ +import { defineConfig } from 'vite'; +import react from '@vitejs/plugin-react'; + +export default defineConfig({ + plugins: [react()], + server: { + host: '127.0.0.1', + strictPort: false, + }, +}); diff --git a/tests/framework-fixtures/vite8-react-brand-fidelity/fixture.json b/tests/framework-fixtures/vite8-react-brand-fidelity/fixture.json new file mode 100644 index 000000000..6cd9f1fa9 --- /dev/null +++ b/tests/framework-fixtures/vite8-react-brand-fidelity/fixture.json @@ -0,0 +1,37 @@ +{ + "name": "Vite 8 + React + brand fidelity", + "config": { + "files": ["index.html"], + "insertBefore": "", + "commentSyntax": "html" + }, + "sourceFiles": ["index.html", "src/App.jsx", "src/main.jsx", "src/styles.css", "vite.config.js"], + "generatedFiles": [], + "wrapCases": [ + { + "name": "wraps the benchmark offer card in source JSX", + "args": { "classes": "offer-card", "tag": "article", "text": "Field Notes" }, + "expectedFile": "src/App.jsx" + } + ], + "runtime": { + "styling": "plain-css", + "install": ["npm", "install", "--no-audit", "--no-fund", "--loglevel=error"], + "devCommand": ["npx", "vite", "--host", "127.0.0.1"], + "readyPattern": "Local:\\s+https?://[^:]+:(\\d+)", + "readyTimeoutMs": 120000, + "pickSelector": "article.offer-card", + "acceptedSourcePattern": "]*(class|className)=\"[^\"]*\\boffer-card\\b[^\"]*\"", + "steer": { + "message": "steer-e2e mark offer", + "target": { "classes": "offer-card", "tag": "article" }, + "expectSelector": "article.offer-card[data-impeccable-steer=\"e2e\"]", + "expectSourceContains": "data-impeccable-steer=\"e2e\"", + "sourceFile": "src/App.jsx" + }, + "probe": { + "expectLiveInit": true, + "expectConsoleClean": true + } + } +} diff --git a/tests/framework-fixtures/vite8-react-brand-fidelity/gitignore.txt b/tests/framework-fixtures/vite8-react-brand-fidelity/gitignore.txt new file mode 100644 index 000000000..4cf6bdb4f --- /dev/null +++ b/tests/framework-fixtures/vite8-react-brand-fidelity/gitignore.txt @@ -0,0 +1,3 @@ +node_modules +dist +.impeccable diff --git a/tests/live-benchmark.test.mjs b/tests/live-benchmark.test.mjs new file mode 100644 index 000000000..10dd4fc4e --- /dev/null +++ b/tests/live-benchmark.test.mjs @@ -0,0 +1,165 @@ +import assert from 'node:assert/strict'; +import { describe, it } from 'node:test'; + +import { + assembleSplitProgressiveOutput, + buildInteractionRun, + compareModelBackedReports, + createTraceRecorder, + durationBetween, + summarizeRuns, +} from '../scripts/lib/live-benchmark.mjs'; + +describe('live benchmark metrics', () => { + it('keeps published progressive CSS byte-stable and carries deferred params', () => { + const firstCss = '@scope ([data-impeccable-variant="1"]) { .offer { color: red; } }'; + const laterCss = [ + '@scope ([data-impeccable-variant="2"]) { .offer { color: green; } }', + '@scope ([data-impeccable-variant="3"]) { .offer { color: blue; } }', + ].join('\n'); + const firstVariant = { innerHtml: '
One
', params: [] }; + const deferredParams = [{ name: 'density', type: 'range', min: 0, max: 1, default: 0.5 }]; + const assembled = assembleSplitProgressiveOutput( + { scopedCss: firstCss, variants: [firstVariant] }, + { + scopedCss: laterCss, + variants: [ + { innerHtml: firstVariant.innerHtml, params: deferredParams }, + { innerHtml: '
Two
', params: [] }, + { innerHtml: '
Three
', params: [] }, + ], + }, + ); + + assert.equal(assembled.scopedCss, `${firstCss}\n${laterCss}`); + assert.equal(assembled.scopedCss.slice(0, firstCss.length), firstCss); + assert.equal(assembled.variants[0].innerHtml, firstVariant.innerHtml); + assert.equal(assembled.variants[0].params, deferredParams); + }); + + it('rejects tail CSS that would reproduce published_variant_css_changed', () => { + const first = { + scopedCss: '@scope ([data-impeccable-variant="1"]) { .offer { color: red; } }', + variants: [{ innerHtml: '
One
', params: [] }], + }; + const conflictingTail = { + scopedCss: [ + '@scope ([data-impeccable-variant="1"]) { .offer { color: purple; } }', + '@scope ([data-impeccable-variant="2"]) { .offer { color: green; } }', + ].join('\n'), + variants: [ + { innerHtml: first.variants[0].innerHtml, params: [] }, + { innerHtml: '
Two
', params: [] }, + ], + }; + + assert.throws( + () => assembleSplitProgressiveOutput(first, conflictingTail), + /must not repeat or conflict with published variant 1 CSS/, + ); + }); + + it('separates model generation from Impeccable overhead', () => { + const events = [ + { name: 'ui.go.start', at: 100, iteration: 1 }, + { name: 'browser.generate_post', at: 108, id: 'abc', hasScreenshotPath: false, commentCount: 0, strokeCount: 0 }, + { name: 'agent.event.received', at: 110, id: 'abc', type: 'generate' }, + { name: 'agent.scaffold.start', at: 112, id: 'abc' }, + { name: 'agent.scaffold.end', at: 132, id: 'abc' }, + { name: 'agent.generate.start', at: 132, id: 'abc' }, + { name: 'agent.generate.first_ready', at: 1132, id: 'abc' }, + { name: 'agent.generate.end', at: 1132, id: 'abc' }, + { name: 'agent.write.start', at: 1132, id: 'abc' }, + { name: 'agent.write.end', at: 1142, id: 'abc' }, + { name: 'agent.reply.start', at: 1142, id: 'abc' }, + { name: 'agent.reply.end', at: 1147, id: 'abc' }, + { name: 'browser.first_variant', at: 1200, iteration: 1 }, + { name: 'browser.all_variants', at: 1200, iteration: 1 }, + ]; + + const run = buildInteractionRun(events, { + iteration: 1, + scenario: 'plain', + goStartedAt: 100, + browserTiming: { goAt: 50, generateAt: 52.5 }, + }); + assert.equal(run.goToFirstVariantMs, 1094.5); + assert.equal(run.browserPreparationMs, 8); + assert.equal(run.browserDispatchMs, 2.5); + assert.equal(run.automationClickMs, 5.5); + assert.deepEqual(run.annotationEvidence, { screenshotPath: false, comments: 0, strokes: 0 }); + assert.equal(run.serverPickupMs, 2); + assert.equal(run.generationMs, 1000); + assert.equal(run.impeccableOverheadMs, 94.5); + assert.equal(run.deliveryGapMs, 0); + assert.equal(run.scaffoldMs, 20); + }); + + it('reports interpolated medians and p95 values', () => { + const summary = summarizeRuns([ + { goToFirstVariantMs: 100, generationMs: 70 }, + { goToFirstVariantMs: 200, generationMs: 140 }, + { goToFirstVariantMs: 300, generationMs: 210 }, + ]); + assert.equal(summary.metrics.goToFirstVariantMs.median, 200); + assert.equal(summary.metrics.goToFirstVariantMs.p95, 290); + }); + + it('records monotonic trace events and returns null for missing boundaries', () => { + let now = 0; + const recorder = createTraceRecorder(() => ++now); + recorder.trace('start'); + recorder.trace('end'); + assert.equal(durationBetween(recorder.events, 'start', 'end'), 1); + assert.equal(durationBetween(recorder.events, 'missing', 'end'), null); + }); + + it('proves model-backed first-reviewable thresholds with comparable reports', () => { + const atomic = modelReport('atomic', 1000, 1200, 1400, 1500); + const progressive = modelReport('progressive', 500, 700, 1450, 1550); + const comparison = compareModelBackedReports(atomic, progressive); + assert.equal(comparison.passed, true); + assert.equal(comparison.target.medianImprovement, 0.5); + assert.equal(comparison.target.p95Improvement, 0.4167); + }); + + it('rejects fake, simulated, and mismatched model reports', () => { + const atomic = modelReport('atomic', 1000, 1200, 1400, 1500); + const progressive = modelReport('progressive', 500, 700, 1450, 1550); + assert.throws( + () => compareModelBackedReports({ ...atomic, benchmark: { ...atomic.benchmark, agent: 'fake' } }, progressive), + /model-backed/, + ); + assert.throws( + () => compareModelBackedReports(atomic, { ...progressive, benchmark: { ...progressive.benchmark, simulation: { remainingGenerationMs: 1 } } }), + /simulated latency/, + ); + assert.throws( + () => compareModelBackedReports(atomic, { ...progressive, benchmark: { ...progressive.benchmark, model: 'other-model' } }), + /benchmark mismatch for model/, + ); + }); +}); + +function modelReport(delivery, firstMedian, firstP95, allMedian, allP95) { + return { + benchmark: { + fixture: 'vite8-react-plain', + agent: 'llm', + provider: 'anthropic', + model: 'claude-haiku-4-5', + scenario: 'plain', + variants: 3, + delivery, + promptMode: 'synthetic-element-contract', + simulation: null, + }, + summary: { + count: 5, + metrics: { + goToFirstVariantMs: { median: firstMedian, p95: firstP95 }, + goToAllVariantsMs: { median: allMedian, p95: allP95 }, + }, + }, + }; +} diff --git a/tests/live-provider-benchmark.test.mjs b/tests/live-provider-benchmark.test.mjs new file mode 100644 index 000000000..354b4eda5 --- /dev/null +++ b/tests/live-provider-benchmark.test.mjs @@ -0,0 +1,115 @@ +import assert from 'node:assert/strict'; +import { describe, it } from 'node:test'; + +import { + STRATEGIES, + assembleProgressiveOutput, + applyRuntimeSourceScore, + estimateCostUsd, + scoreVariantOutput, + summarizeProviderRuns, + validateAcceptedCleanup, +} from '../scripts/lib/live-provider-benchmark.mjs'; + +const VARIANT = [ + '
', + '
', + '

Quarterly print edition

', + '

Field Notes

', + '

Four routes, annotated maps, and practical details for unhurried weekends.

', + '
', + 'Reserve issue eight', + '
', +].join(''); + +const GOOD_OUTPUT = { + scopedCss: [ + '@scope ([data-impeccable-variant="1"]) {', + ' :scope > .offer-card { background: var(--color-paper-deep); color: var(--color-ink); gap: var(--space-3); }', + ' :scope .offer-card__eyebrow { color: var(--color-moss); }', + '}', + ].join('\n'), + variants: [{ innerHtml: VARIANT, params: [] }], +}; + +describe('cross-provider Live benchmark', () => { + it('defines the control, progressive, compact, and parallel candidates', () => { + assert.deepEqual(Object.keys(STRATEGIES), [ + 'atomic-full', + 'progressive-full', + 'progressive-compact', + 'parallel-compact', + ]); + }); + + it('assembles progressive output without asking the tail call to reproduce variant 1', () => { + const first = { + scopedCss: '@scope ([data-impeccable-variant="1"]) { .first { color: var(--color-ink); } }', + variants: [{ innerHtml: VARIANT, params: [] }], + }; + const remaining = { + scopedCss: [ + '@scope ([data-impeccable-variant="1"]) { .second { color: var(--color-moss); } }', + '@scope ([data-impeccable-variant="2"]) { .third { color: var(--color-brass); } }', + ].join('\n'), + variants: [{ innerHtml: `${VARIANT} ` }, { innerHtml: `${VARIANT} ` }], + }; + const assembled = assembleProgressiveOutput(first, remaining); + assert.equal(assembled.variants[0], first.variants[0]); + assert.ok(assembled.scopedCss.startsWith(first.scopedCss)); + assert.match(assembled.scopedCss, /data-impeccable-variant="2"[^]*second/); + assert.match(assembled.scopedCss, /data-impeccable-variant="3"[^]*third/); + }); + + it('passes on-brand, token-driven, copy-preserving component output', () => { + const score = scoreVariantOutput(GOOD_OUTPUT); + assert.equal(score.brandFidelity, 1); + assert.equal(score.componentFidelity, 1); + assert.equal(score.copyFidelity, 1); + assert.equal(score.sourceValidity, 1); + assert.ok(score.tokenFidelity >= 0.75); + assert.equal(score.passed, true); + }); + + it('rejects off-brand raw colors, missing component parts, and changed copy', () => { + const score = scoreVariantOutput({ + scopedCss: '.offer-card { color: #ff00ff; background: linear-gradient(red, blue); box-shadow: 0 0 20px cyan; }', + variants: [{ innerHtml: '
Different sales copy
' }], + }); + assert.ok(score.brandFidelity < 0.75); + assert.ok(score.componentFidelity < 0.75); + assert.equal(score.copyFidelity, 0); + assert.equal(score.passed, false); + }); + + it('requires the accepted source to build and lose every Live marker', () => { + const cleanSource = `export default function Card(){return (${VARIANT.replaceAll('class=', 'className=')});}`; + const cleanup = validateAcceptedCleanup({ source: cleanSource, browserClean: true, buildPassed: true }); + assert.equal(cleanup.passed, true); + + const dirty = validateAcceptedCleanup({ + source: `${cleanSource}\n{/* impeccable-carbonize-start test */}`, + browserClean: true, + buildPassed: true, + }); + assert.equal(dirty.markerFree, false); + assert.equal(dirty.passed, false); + assert.equal(applyRuntimeSourceScore(scoreVariantOutput(GOOD_OUTPUT), dirty).passed, false); + }); + + it('estimates cached token cost and summarizes latency, quality, and cleanup', () => { + assert.equal(estimateCostUsd( + { inputTokens: 1_000_000, cachedInputTokens: 500_000, outputTokens: 100_000 }, + { input: 3, cachedInput: 0.3, output: 15 }, + ), 3.15); + + const summary = summarizeProviderRuns([ + { firstReviewableMs: 100, allReadyMs: 300, acceptCleanupMs: 20, estimatedCostUsd: 0.1, quality: { ...scoreVariantOutput(GOOD_OUTPUT), sourceValidity: 1 }, cleanup: { passed: true }, passed: true }, + { firstReviewableMs: 200, allReadyMs: 400, acceptCleanupMs: 30, estimatedCostUsd: 0.2, quality: { ...scoreVariantOutput(GOOD_OUTPUT), sourceValidity: 1 }, cleanup: { passed: true }, passed: true }, + ]); + assert.equal(summary.metrics.firstReviewableMs.median, 150); + assert.equal(summary.cleanupPassRate, 1); + assert.equal(summary.gatePassRate, 1); + assert.equal(summary.estimatedCostUsd, 0.3); + }); +}); diff --git a/tests/live-provider-benchmark/README.md b/tests/live-provider-benchmark/README.md new file mode 100644 index 000000000..3159f5fe4 --- /dev/null +++ b/tests/live-provider-benchmark/README.md @@ -0,0 +1,54 @@ +# Live cross-provider benchmark + +This benchmark compares Live variant delivery strategies without confusing model latency with browser/poller overhead. It uses the realistic `vite8-react-brand-fidelity` fixture and scores every output with deterministic gates for: + +- brand fidelity; +- component fidelity; +- CSS-token fidelity; +- exact copy fidelity; +- source/schema validity; +- provider-independent Accept cleanup and production build validity. + +The model matrix and the cleanup control are intentionally separate. Provider generation runs use a fixed synthetic picker event. The cleanup control runs a real Vite/React Live session in Playwright, accepts variant 1, waits for Pick mode, checks that Live markers are gone, and builds the accepted source. This prevents a provider from being blamed for local publisher/poller behavior while retaining a real pipeline safety gate. + +## Commands + +Validate the matrix without API or browser calls: + +```sh +npm run bench:live:providers -- --dry-run +``` + +Run the recommended small matrix and write a report: + +```sh +npm run bench:live:providers -- \ + --strategies atomic-full,progressive-compact,parallel-compact \ + --iterations 1 \ + --output artifacts/live-provider-benchmark.json +``` + +Run only the real Accept/build cleanup control: + +```sh +npm run bench:live:providers -- --cleanup-only --output /tmp/live-cleanup.json +``` + +Arguments accept either `--name=value` or `--name value`. No output file is created unless `--output` is supplied. API keys load, in order, from `--env-file`, the repo `.env`, and `~/code/impeccable-evals/.env`; reports include only key availability, never key values. + +## Strategies + +- `atomic-full`: one call generates all variants with the full Live reference. This is the latency and cost control. +- `progressive-full`: a first-variant call followed by a remaining-directions call, both with full Live context. +- `progressive-compact`: the same split with the stable compact producer contract. Variant 1 and its CSS segment are carried forward byte-for-byte and assembled locally. +- `parallel-compact`: three compact one-variant producers run concurrently. The first valid result is reviewable immediately; centralized assembly remaps the other CSS scopes deterministically. + +The earlier idea of asking the second progressive call to reproduce variant 1 is deliberately excluded. It adds tokens, permits drift, and conflicts with transactional source publication. Deterministic assembly is the production candidate. + +## Interpretation + +A run passes only when overall fidelity is at least `0.90`, every dimension is at least `0.75`, and the real cleanup control passes. One iteration is a smoke matrix, not a statistically stable claim; use at least five iterations before setting a release threshold. + +Cost estimates use provider-reported token counts and standard per-million-token prices recorded on July 11, 2026. Update `PROVIDER_PROFILES` when model pricing changes. Price sources are embedded in every report. + +Latency runs use low effort for Claude Sonnet and GPT, and Gemini 3.1 Flash-Lite's minimal-thinking default. These settings are emitted in provider metadata so a report cannot silently compare different reasoning budgets.
Live event delivery paths by agent harness
PathPickupTradeoffStatus