diff --git a/package.json b/package.json index da12ab1a6..fcb17746b 100644 --- a/package.json +++ b/package.json @@ -70,6 +70,7 @@ "bench:detector:browser": "node scripts/benchmark-detector.mjs --browser", "bench:live": "node scripts/benchmark-live.mjs", "bench:live:providers": "node scripts/benchmark-live-providers.mjs", + "bench:live:codex-quality": "node scripts/benchmark-live-codex-worker.mjs", "audit": "bun audit --audit-level=moderate", "prepack": "cp README.md README.repo.md && cp README.npm.md README.md", "postpack": "cp README.repo.md README.md && rm README.repo.md", diff --git a/scripts/benchmark-live-codex-worker.mjs b/scripts/benchmark-live-codex-worker.mjs new file mode 100644 index 000000000..8f93e4324 --- /dev/null +++ b/scripts/benchmark-live-codex-worker.mjs @@ -0,0 +1,219 @@ +#!/usr/bin/env node + +import { mkdir, readFile, writeFile } from 'node:fs/promises'; +import path from 'node:path'; +import { performance } from 'node:perf_hooks'; +import { fileURLToPath } from 'node:url'; + +import { anthropic } from '@ai-sdk/anthropic'; +import { generateText } from 'ai'; + +import { CodexAppServerClient } from '../skill/scripts/live/codex-app-server-client.mjs'; +import { loadBenchmarkEnv } from './lib/live-provider-benchmark.mjs'; +import { + CODEX_QUALITY_OUTPUT_SCHEMA, + buildCodexQualityPrompt, + buildJudgePrompt, + createCodexQualityTasks, + parseJudgeResult, + scoreCodexQualityOutput, + summarizeCodexQualityRuns, +} from './lib/live-codex-quality-benchmark.mjs'; + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); +const args = parseArgs(process.argv.slice(2)); +const iterations = positiveInteger(args.iterations, 1); +const outputPath = args.output ? path.resolve(ROOT, String(args.output)) : null; +const selectedTaskIds = csv(args.tasks || 'editorial-bolder,operations-polish'); +const selectedProfileIds = csv(args.profiles || 'spark-thin,sol-thin,sol-full'); +const judgeEnabled = args.judge !== false; +const loadedEnv = loadBenchmarkEnv({ repoRoot: ROOT, explicitPath: args.envFile && path.resolve(args.envFile) }); +const skillPath = path.join(ROOT, '.agents', 'skills', 'impeccable', 'SKILL.md'); +const referenceDir = path.join(ROOT, 'skill', 'reference'); +const tasks = createCodexQualityTasks({ repoRoot: ROOT }).filter((task) => selectedTaskIds.includes(task.id)); + +if (tasks.length !== selectedTaskIds.length) throw new Error('unknown task id in --tasks'); + +const client = new CodexAppServerClient({ cwd: ROOT, turnTimeoutMs: positiveInteger(args.timeout, 240_000) }); +await client.connect(); +const models = await client.listModels(); +const profiles = resolveProfiles(selectedProfileIds, models); + +if (args.dryRun) { + const report = { + schemaVersion: 1, + mode: 'dry-run', + iterations, + tasks: tasks.map((task) => ({ id: task.id, action: task.action })), + profiles: profiles.map(publicProfile), + judgeEnabled, + judgeAvailable: Boolean(process.env.ANTHROPIC_API_KEY), + envFilesLoaded: loadedEnv.length, + }; + await client.close(); + await emit(report); + process.exit(0); +} + +if (judgeEnabled && !process.env.ANTHROPIC_API_KEY) { + await client.close(); + throw new Error('ANTHROPIC_API_KEY is required unless --no-judge is passed'); +} + +const runs = []; +try { + for (const profile of profiles) { + for (const task of tasks) { + for (let iteration = 1; iteration <= iterations; iteration += 1) { + process.stderr.write(`[codex-quality] ${profile.id} ${task.id} ${iteration}/${iterations}\n`); + runs.push(await runOne({ client, profile, task, iteration })); + } + } + } +} finally { + await client.close().catch(() => {}); +} + +const report = { + schemaVersion: 1, + generatedAt: new Date().toISOString(), + mode: 'live', + iterations, + judge: judgeEnabled ? { provider: 'anthropic', model: args.judgeModel || 'claude-sonnet-4-6' } : null, + tasks: tasks.map((task) => ({ id: task.id, action: task.action, brief: task.brief })), + profiles: profiles.map((profile) => ({ + ...publicProfile(profile), + summary: summarizeCodexQualityRuns(runs.filter((run) => run.profile === profile.id)), + })), + runs, +}; +await emit(report); +process.exitCode = runs.every((run) => run.passed) ? 0 : 1; + +async function runOne({ client: appServer, profile, task, iteration }) { + let thread = null; + try { + const actionReference = await readFile(path.join(referenceDir, `${task.action}.md`), 'utf-8'); + thread = await appServer.startDedicatedThread({ + model: profile.model, + cwd: ROOT, + approvalPolicy: 'never', + sandbox: 'read-only', + ephemeral: true, + serviceName: `impeccable_live_quality_${profile.id}`, + baseInstructions: profile.fullContext + ? 'You are a read-only Impeccable frontend implementation worker. The supervisor supplies resolved context and owns all writes. Return only schema-valid JSON.' + : 'You are a dedicated Impeccable Live variant producer. Do not use tools or inspect files. Return only schema-valid JSON. Preserve copy, component contracts, accessibility, and supplied tokens.', + }); + const prompt = buildCodexQualityPrompt(task, { actionReference, fullContext: profile.fullContext }); + const input = profile.fullContext + ? [{ type: 'skill', name: 'impeccable', path: skillPath }, { type: 'text', text: prompt }] + : [{ type: 'text', text: prompt }]; + const startedAt = performance.now(); + const result = await appServer.startTurn({ + threadId: thread.id, + input, + cwd: ROOT, + model: profile.model, + effort: profile.effort, + summary: 'none', + approvalPolicy: 'never', + sandboxPolicy: { type: 'readOnly' }, + outputSchema: CODEX_QUALITY_OUTPUT_SCHEMA, + }); + const durationMs = Math.round(performance.now() - startedAt); + const output = JSON.parse(result.message); + const deterministic = scoreCodexQualityOutput(task, output); + const judge = judgeEnabled ? await judgeOutput(task, output) : null; + return { + profile: profile.id, + task: task.id, + iteration, + model: profile.model, + effort: profile.effort, + fullContext: profile.fullContext, + durationMs, + deterministic, + judge, + passed: deterministic.passed && (!judge || judge.passed), + output, + }; + } catch (error) { + return { + profile: profile.id, + task: task.id, + iteration, + model: profile.model, + effort: profile.effort, + fullContext: profile.fullContext, + error: String(error?.stack || error), + passed: false, + }; + } finally { + if (thread) await appServer.archiveThread(thread.id).catch(() => {}); + } +} + +async function judgeOutput(task, output) { + const response = await generateText({ + model: anthropic(args.judgeModel || 'claude-sonnet-4-6'), + system: 'Be strict, concrete, and independent. Return the requested JSON object only.', + prompt: buildJudgePrompt(task, output), + maxOutputTokens: 800, + }); + return parseJudgeResult(response.text); +} + +function resolveProfiles(ids, models) { + const visible = models.filter((model) => model && !model.hidden); + const find = (patterns) => visible.find((model) => patterns.every((pattern) => pattern.test(`${model.id || ''} ${model.model || ''}`))); + const spark = find([/spark/i]); + const sol = find([/5\.6/i, /sol/i]) || visible.find((model) => model.isDefault); + const catalog = { + 'spark-thin': { id: 'spark-thin', model: spark?.model || spark?.id, effort: spark?.defaultReasoningEffort || 'high', fullContext: false }, + 'sol-thin': { id: 'sol-thin', model: sol?.model || sol?.id, effort: 'medium', fullContext: false }, + 'sol-full': { id: 'sol-full', model: sol?.model || sol?.id, effort: 'medium', fullContext: true }, + }; + return ids.map((id) => { + const profile = catalog[id]; + if (!profile) throw new Error(`unknown profile ${id}`); + if (!profile.model) throw new Error(`model unavailable for ${id}`); + return profile; + }); +} + +function publicProfile(profile) { + return { id: profile.id, model: profile.model, effort: profile.effort, fullContext: profile.fullContext }; +} + +async function emit(report) { + if (outputPath) { + await mkdir(path.dirname(outputPath), { recursive: true }); + await writeFile(outputPath, JSON.stringify(report, null, 2) + '\n'); + } + process.stdout.write(JSON.stringify(report, null, 2) + '\n'); +} + +function parseArgs(values) { + const result = {}; + for (let index = 0; index < values.length; index += 1) { + const value = values[index]; + if (value === '--dry-run') result.dryRun = true; + else if (value === '--no-judge') result.judge = false; + else if (value.startsWith('--')) { + const [rawKey, inline] = value.slice(2).split('=', 2); + const key = rawKey.replace(/-([a-z])/g, (_, letter) => letter.toUpperCase()); + result[key] = inline ?? values[++index]; + } + } + return result; +} + +function csv(value) { + return String(value).split(',').map((item) => item.trim()).filter(Boolean); +} + +function positiveInteger(value, fallback) { + const parsed = Number.parseInt(value, 10); + return Number.isInteger(parsed) && parsed > 0 ? parsed : fallback; +} diff --git a/scripts/lib/live-codex-quality-benchmark.mjs b/scripts/lib/live-codex-quality-benchmark.mjs new file mode 100644 index 000000000..e47e641d1 --- /dev/null +++ b/scripts/lib/live-codex-quality-benchmark.mjs @@ -0,0 +1,270 @@ +import { readFileSync } from 'node:fs'; +import path from 'node:path'; + +export const CODEX_QUALITY_OUTPUT_SCHEMA = Object.freeze({ + type: 'object', + properties: { + files: { + type: 'array', + minItems: 2, + maxItems: 2, + items: { + type: 'object', + properties: { + path: { type: 'string', enum: ['src/App.jsx', 'src/styles.css'] }, + content: { type: 'string', minLength: 1 }, + }, + required: ['path', 'content'], + additionalProperties: false, + }, + }, + }, + required: ['files'], + additionalProperties: false, +}); + +const EDITORIAL_PRODUCT = `# Northstar Field Journal + +An independent quarterly field guide for design-conscious weekend walkers. Readers value practical detail, editorial restraint, and objects worth keeping. The offer card should make issue eight feel collectible without becoming luxurious or loud.`; + +const EDITORIAL_DESIGN = `# Design system + +- Warm paper, dark ink, moss, and brass only. +- Georgia display type with a restrained sans body. +- Editorial, practical, quiet, and tactile. +- Reuse the existing CSS custom properties. Do not add colors, fonts, gradients, shadows, glow, glass, or decorative effects. +- Square, rule-led compositions are preferred to card stacks and rounded containers.`; + +const OPERATIONS_APP = `function Metric({ label, value, detail, tone = 'neutral' }) { + return ( +
+

{label}

+ {value} +

{detail}

+
+ ); +} + +export default function App() { + return ( +
+
+
+

Monday, 14 July

+

Fulfillment overview

+

Monitor the work that can put today’s dispatch at risk.

+
+ +
+ +
+ + + +
+ +
+
+
+

Priority queue

+

Needs attention

+
+ +
+ + + + + + + +
DispatchDestinationOwnerStatusDue
DP-2048PortlandUnassignedBlocked09:30
DP-2051OaklandM. ChenAt risk10:15
DP-2057SeattleA. SinghReview11:00
+
+
+ ); +}`; + +const OPERATIONS_CSS = `:root { + --canvas: #f5f6f7; + --surface: #ffffff; + --surface-subtle: #eef0f2; + --ink: #17202a; + --ink-muted: #66717d; + --line: #d8dde2; + --accent: #176b5b; + --positive: #176b5b; + --warning: #925f09; + --critical: #a83d32; + --space-1: 0.375rem; + --space-2: 0.75rem; + --space-3: 1rem; + --space-4: 1.5rem; + --space-5: 2rem; + --radius: 0.375rem; + font-family: Inter, ui-sans-serif, system-ui, sans-serif; +} +* { box-sizing: border-box; } +body { margin: 0; background: var(--canvas); color: var(--ink); } +button { font: inherit; } +.workspace { width: min(80rem, calc(100% - 2rem)); margin: 0 auto; padding: var(--space-5) 0; } +.workspace__header, .queue__heading { display: flex; align-items: end; justify-content: space-between; gap: var(--space-4); } +.eyebrow, .summary, .metric__label, .metric__detail { margin: 0; color: var(--ink-muted); } +h1 { margin: var(--space-1) 0; font-size: 2rem; } +h2 { margin: var(--space-1) 0 0; font-size: 1.25rem; } +.button { min-height: 2.5rem; padding: 0 var(--space-3); border: 1px solid var(--line); border-radius: var(--radius); background: var(--surface); color: var(--ink); font-weight: 650; } +.button--primary { border-color: var(--accent); background: var(--accent); color: white; } +.button--quiet { background: transparent; } +.metrics { display: grid; grid-template-columns: repeat(3, 1fr); gap: var(--space-3); margin: var(--space-5) 0; } +.metric, .queue { border: 1px solid var(--line); border-radius: var(--radius); background: var(--surface); } +.metric { padding: var(--space-4); } +.metric__value { display: block; margin: var(--space-2) 0; font-size: 2rem; } +.metric--warning { border-top: 0.25rem solid var(--warning); } +.metric--critical { border-top: 0.25rem solid var(--critical); } +.metric--positive { border-top: 0.25rem solid var(--positive); } +.queue { overflow: hidden; } +.queue__heading { padding: var(--space-4); border-bottom: 1px solid var(--line); } +table { width: 100%; border-collapse: collapse; } +th, td { padding: var(--space-3) var(--space-4); border-bottom: 1px solid var(--line); text-align: left; } +th { background: var(--surface-subtle); color: var(--ink-muted); font-size: 0.75rem; text-transform: uppercase; letter-spacing: 0.05em; } +.status { display: inline-flex; padding: var(--space-1) var(--space-2); border-radius: var(--radius); background: var(--surface-subtle); font-weight: 650; } +.status--warning { color: var(--warning); } +.status--critical { color: var(--critical); } +@media (max-width: 44rem) { + .workspace__header { align-items: stretch; flex-direction: column; } + .metrics { grid-template-columns: 1fr; } + .queue { overflow-x: auto; } +}`; + +export function createCodexQualityTasks({ repoRoot }) { + const fixtureDir = path.join(repoRoot, 'tests', 'framework-fixtures', 'vite8-react-brand-fidelity', 'files', 'src'); + return [ + { + id: 'editorial-bolder', + action: 'bolder', + brief: 'Make the Field Notes offer card materially bolder. Keep it unmistakably Northstar: amplify hierarchy, proportion, and composition inside the existing design system. Preserve every word and the ActionLink component.', + product: EDITORIAL_PRODUCT, + design: EDITORIAL_DESIGN, + files: { + 'src/App.jsx': readFileSync(path.join(fixtureDir, 'App.jsx'), 'utf-8'), + 'src/styles.css': readFileSync(path.join(fixtureDir, 'styles.css'), 'utf-8'), + }, + requiredCopy: ['Quarterly print edition', 'Field Notes', 'Four routes, annotated maps, and practical details for unhurried weekends.', 'Reserve issue eight'], + requiredSource: ['function ActionLink', 'Reserve issue eight', 'aria-labelledby="field-notes-title"'], + requiredTokens: ['--color-paper', '--color-paper-deep', '--color-ink', '--color-moss', '--color-brass', '--font-display', '--font-body'], + forbidden: [/gradient\s*\(/i, /box-shadow\s*:/i, /filter\s*:\s*blur/i, /#[0-9a-f]{3,8}\b/gi], + judgeFocus: 'Is the selected offer materially more decisive through hierarchy/proportion/composition, while remaining restrained editorial design rather than generic AI boldness?', + }, + { + id: 'operations-polish', + action: 'polish', + brief: 'Polish this fulfillment dashboard to flagship quality. Improve hierarchy, scanning, density, alignment, interaction states, responsive behavior, and accessibility. Keep the existing information architecture, components, terminology, and token palette.', + product: '# Relay\n\nAn operations workspace for fulfillment leads. The dashboard must support rapid scanning under time pressure; calm precision matters more than personality or visual novelty.', + design: '# Relay design system\n\nCompact, neutral, table-first application UI. Use existing tokens and components. Status color communicates meaning only. Avoid gradients, decorative shadows, oversized display type, rounded-card proliferation, and invented navigation.', + files: { 'src/App.jsx': OPERATIONS_APP, 'src/styles.css': OPERATIONS_CSS }, + requiredCopy: ['Fulfillment overview', 'Create dispatch', 'Ready', 'At risk', 'Blocked', 'Needs attention', 'View all 19', 'DP-2048', 'DP-2051', 'DP-2057'], + requiredSource: ['function Metric', '', 'aria-labelledby="queue-title"'], + requiredTokens: ['--canvas', '--surface', '--ink', '--ink-muted', '--line', '--accent', '--positive', '--warning', '--critical'], + forbidden: [/gradient\s*\(/i, /box-shadow\s*:/i, /backdrop-filter/i, /border-radius\s*:\s*(?:1|2|3|4|5|6|7|8|9)rem/i], + judgeFocus: 'Is this a materially more polished, efficient operations surface, with excellent scan hierarchy and interaction detail, without changing its product model or turning it into a decorative dashboard?', + }, + ]; +} + +export function buildCodexQualityPrompt(task, { actionReference = '', fullContext = false } = {}) { + return [ + `Impeccable Live task: /${task.action}`, + task.brief, + '', + 'Return exactly the two complete revised files required by the output schema. Do not explain the answer.', + 'This is an automated one-shot task: do not ask questions. Preserve visible copy and functional component contracts.', + fullContext ? 'The Impeccable skill is attached. Its Setup context has already been resolved and is included below; do not rerun setup.' : '', + '', + '', task.product, '', + '', task.design, '', + '', actionReference, '', + '', JSON.stringify(task.files, null, 2), '', + ].filter((line) => line !== '').join('\n'); +} + +export function scoreCodexQualityOutput(task, output) { + const files = Array.isArray(output?.files) ? output.files : []; + const byPath = Object.fromEntries(files.map((file) => [file?.path, String(file?.content || '')])); + const combined = Object.values(byPath).join('\n'); + const source = byPath['src/App.jsx'] || ''; + const css = byPath['src/styles.css'] || ''; + const checks = { + exactFiles: files.length === 2 && Boolean(source) && Boolean(css), + sourceChanged: source !== task.files['src/App.jsx'], + stylesChanged: css !== task.files['src/styles.css'], + copyPreserved: task.requiredCopy.every((value) => combined.includes(value)), + contractsPreserved: task.requiredSource.every((value) => source.includes(value)), + tokensPreserved: task.requiredTokens.every((value) => css.includes(value)), + noForbiddenDrift: task.forbidden.every((pattern) => { + pattern.lastIndex = 0; + const outputMatches = combined.match(pattern) || []; + pattern.lastIndex = 0; + const inputMatches = Object.values(task.files).join('\n').match(pattern) || []; + return outputMatches.length <= inputMatches.length; + }), + }; + return { + passed: Object.values(checks).every(Boolean), + checks, + }; +} + +export function buildJudgePrompt(task, output) { + return [ + 'You are an exacting independent frontend design reviewer. Score the revised code, not the prose around it.', + 'Return JSON only with integer scores from 1-10 for commandFidelity, brandAndSystemFidelity, frontendQuality, and taskCompletion; plus criticalFailure (boolean) and summary (one short sentence).', + 'A score of 7 means clearly shippable and materially improved. Penalize generic AI aesthetics, token drift, invented content, component destruction, and superficial changes.', + '', + `TASK: /${task.action} — ${task.brief}`, + `REVIEW FOCUS: ${task.judgeFocus}`, + '', task.product, '', + '', task.design, '', + '', JSON.stringify(task.files, null, 2), '', + '', JSON.stringify(output?.files || [], null, 2), '', + ].join('\n'); +} + +export function parseJudgeResult(text) { + const match = String(text || '').match(/\{[\s\S]*\}/); + if (!match) throw new Error('judge returned no JSON object'); + const result = JSON.parse(match[0]); + const keys = ['commandFidelity', 'brandAndSystemFidelity', 'frontendQuality', 'taskCompletion']; + const scores = Object.fromEntries(keys.map((key) => [key, Number(result[key])])); + const passed = keys.every((key) => Number.isInteger(scores[key]) && scores[key] >= 7) + && result.criticalFailure !== true; + return { ...result, ...scores, passed }; +} + +export function summarizeCodexQualityRuns(runs) { + const finished = runs.filter((run) => !run.error); + const latencies = finished.map((run) => run.durationMs).sort((a, b) => a - b); + const judged = finished.filter((run) => run.judge); + const scoreKeys = ['commandFidelity', 'brandAndSystemFidelity', 'frontendQuality', 'taskCompletion']; + return { + runs: runs.length, + passed: finished.filter((run) => run.passed).length, + medianDurationMs: percentile(latencies, 0.5), + p95DurationMs: percentile(latencies, 0.95), + averageJudgeScores: Object.fromEntries(scoreKeys.map((key) => [ + key, + judged.length ? round(judged.reduce((sum, run) => sum + run.judge[key], 0) / judged.length) : null, + ])), + }; +} + +function percentile(values, quantile) { + if (values.length === 0) return null; + const index = (values.length - 1) * quantile; + const lower = Math.floor(index); + const upper = Math.ceil(index); + if (lower === upper) return round(values[lower]); + return round(values[lower] + (values[upper] - values[lower]) * (index - lower)); +} + +function round(value) { + return Math.round(value * 100) / 100; +} diff --git a/tests/live-codex-quality-benchmark.test.mjs b/tests/live-codex-quality-benchmark.test.mjs new file mode 100644 index 000000000..f978e87ca --- /dev/null +++ b/tests/live-codex-quality-benchmark.test.mjs @@ -0,0 +1,57 @@ +import assert from 'node:assert/strict'; +import path from 'node:path'; +import { describe, it } from 'node:test'; + +import { + buildCodexQualityPrompt, + createCodexQualityTasks, + parseJudgeResult, + scoreCodexQualityOutput, + summarizeCodexQualityRuns, +} from '../scripts/lib/live-codex-quality-benchmark.mjs'; + +const repoRoot = path.resolve(import.meta.dirname, '..'); +const tasks = createCodexQualityTasks({ repoRoot }); + +describe('Codex Live quality benchmark', () => { + it('covers full bolder and polish tasks with project and action context', () => { + assert.deepEqual(tasks.map((task) => `${task.action}:${task.id}`), [ + 'bolder:editorial-bolder', + 'polish:operations-polish', + ]); + const prompt = buildCodexQualityPrompt(tasks[0], { actionReference: 'BOLDER', fullContext: true }); + assert.match(prompt, /Impeccable skill is attached/); + assert.match(prompt, //); + assert.match(prompt, //); + assert.match(prompt, /BOLDER/); + assert.match(prompt, /src\/App\.jsx/); + }); + + it('rejects no-op, contract-breaking, and design-system-drifting output', () => { + const task = tasks[0]; + const noOp = { files: Object.entries(task.files).map(([filePath, content]) => ({ path: filePath, content })) }; + assert.equal(scoreCodexQualityOutput(task, noOp).passed, false); + + const drift = { + files: [ + { path: 'src/App.jsx', content: task.files['src/App.jsx'].replace('Field Notes', 'Neon Notes') }, + { path: 'src/styles.css', content: `${task.files['src/styles.css']}\n.offer-card { background: linear-gradient(red, blue); }` }, + ], + }; + const score = scoreCodexQualityOutput(task, drift); + assert.equal(score.checks.copyPreserved, false); + assert.equal(score.checks.noForbiddenDrift, false); + }); + + it('parses strict judge results and summarizes latency and quality', () => { + const judge = parseJudgeResult('{"commandFidelity":8,"brandAndSystemFidelity":9,"frontendQuality":7,"taskCompletion":8,"criticalFailure":false,"summary":"Good."}'); + assert.equal(judge.passed, true); + const summary = summarizeCodexQualityRuns([ + { durationMs: 100, passed: true, judge }, + { durationMs: 300, passed: false, judge: { ...judge, frontendQuality: 6 } }, + ]); + assert.equal(summary.medianDurationMs, 200); + assert.equal(summary.passed, 1); + assert.equal(summary.averageJudgeScores.frontendQuality, 6.5); + }); +});