Test advice and documentation outcomes rather than reference coverage

Keep completion, consent, documentation evidence, and new-world artifact requirements strict. Record reference-only omissions as diagnostics in the scoped advice and no-change fixtures.

AI assistance: Codex, under maintainer direction.
This commit is contained in:
Paul Bakaus
2026-09-07 18:33:35 -07:00
parent 2ef885efcf
commit 5b5fdf040f
6 changed files with 158 additions and 28 deletions
+47 -2
View File
@@ -4,10 +4,55 @@ import fs from 'node:fs';
import path from 'node:path';
import { MockLanguageModelV3 } from 'ai/test';
import { prepareWorkspace, cleanupWorkspace, makeTools, runTurn, fileLoaded, SKILL_BODY } from './skill-behavior/harness.mjs';
import { assertPlanningFallbackWarning, assertNewWorkLifecycle } from './skill-behavior/assertions.mjs';
import { assertPlanningFallbackWarning, assertNewWorkLifecycle, assertWorkflowAdvice, assertCommandComparison, missingReferences } from './skill-behavior/assertions.mjs';
import { CASE_STUDY_ANSWER } from './skill-behavior/fixtures.mjs';
import { sourceHash as hashSources } from './skill-workflow/source-hash.mjs';
import { assertCompleted, assertFreshCaptures } from './skill-workflow/assertions.mjs';
import { assertCompleted, assertFreshCaptures, assertNoChangeDocumentation } from './skill-workflow/assertions.mjs';
it('advice outcomes do not depend on opening every reference, but keep consent and prerequisite gates', () => {
const trace = { toolCalls: [], writePaths: [], questionCalls: [], bashCommands: [] };
const advice = 'For index.html, start with init to capture context and document to record the existing identity.';
const comparison = 'Critique is an assessment. Polish makes fixes. Critique is optional; polish can run directly.';
const check = (text, observation = trace) => assertCommandComparison(observation, text);
assert.doesNotThrow(() => assertWorkflowAdvice(trace, advice, { missingContext: true }));
assert.doesNotThrow(() => check(comparison));
assert.doesNotThrow(() => check("Critique reviews the surface. Polish refines it. Critique isn't required before polish."));
assert.doesNotThrow(() => check('Critique gives a report. Polish edits the surface independently, without a critique.'));
assert.deepEqual(missingReferences(trace, ['reference/critique.md']), ['reference/critique.md']);
for (const wrong of ['You must run critique before polish.', 'Critique is required before polish.', 'Polish requires a critique.']) {
assert.throws(() => check(`${comparison} ${wrong}`), /invent a critique prerequisite/);
}
for (const wrong of ['You must run init before polishing.', 'You need to document before refinement.', 'Polish requires PRODUCT.md.']) {
assert.throws(() => assertWorkflowAdvice(trace, `${advice} ${wrong}`, { missingContext: true }), /mandatory prerequisite/);
}
assert.throws(() => check(''), /advice must reach/);
assert.throws(() => check('I loaded critique.md and polish.md.'), /explain critique/);
for (const mutation of ['index.html', '.impeccable/critique/report.md']) {
assert.throws(() => check(comparison, { ...trace, toolCalls: [{ mutatedPaths: [mutation] }] }), /must not edit/);
}
assert.throws(() => check(comparison, { ...trace, writePaths: ['DESIGN.md'] }), /write tool/);
assert.throws(() => check(comparison, { ...trace, questionCalls: [{}] }), /interview/);
assert.throws(() => check(comparison, { ...trace, bashCommands: ['impeccable detect index.html'] }), /menu scans/);
});
it('an unchanged documentation outcome needs real reads and a supported report, not a wrapper filename', () => {
const files = ['reference/document.md', 'index.html', 'DESIGN.md'];
const toolCalls = files.map((file) => ({ loadedFiles: [file] }));
const result = { outcome: 'complete', trace: { toolCalls }, text: 'No changes: index.html matches DESIGN.md: system-ui, 65ch, #0645ad.' };
const options = { target: 'index.html', evidence: [/system-ui/, /65ch/, /#0645ad/] };
const check = (value) => assertNoChangeDocumentation(value, options);
assert.doesNotThrow(() => check(result));
assert.deepEqual(missingReferences(result.trace, ['degraded/documenter.md']), ['degraded/documenter.md']);
for (const missing of files) {
assert.throws(() => check({ ...result, trace: { toolCalls: toolCalls.filter((call) => !call.loadedFiles.includes(missing)) } }), /inspect the actual source/);
}
assert.throws(() => check({ ...result, trace: { toolCalls: [{ input: { path: 'reference/document.md' }, succeeded: false }] } }), /inspect the actual source/);
assert.throws(() => check({ ...result, text: 'No changes: index.html matches DESIGN.md.' }), /report evidence/);
assert.throws(() => check({ ...result, outcome: 'step-budget' }), /did not finish/);
for (const file of ['DESIGN.md', '.impeccable/design.json', 'index.html']) {
assert.throws(() => check({ ...result, trace: { toolCalls: [...toolCalls, { mutatedPaths: [file] }] } }), /must not mutate/);
}
});
it('full workflows reject exhausted budgets and stale or absent visual evidence', () => {
for (const outcome of ['checkpoint', 'step-budget', 'output-limit', 'error']) {
+28 -2
View File
@@ -119,6 +119,32 @@ workflow claim; the reference-loading miss remains visible. No full build was
rerun. Focused Claude routing S3/S4, the ordinary suite, source-first build, and
generated-skill authoring validation passed. Release #782 remains held.
### Outcome assertions (maintainer-approved follow-up)
S16/S17 now require a completed, useful, read-only answer; S17 distinguishes
assessment from implementation and explains that critique is optional before
polish. Explicit invented prerequisites remain failures. Missing routing or
comparison references are TAP diagnostics based on successful content loads,
not failed read attempts. These English-fixture phrase checks are bounded
regression checks, not a comprehensive semantic grader.
The resumed ordinary-extension checkpoint accepts a direct documentation pass
without the degraded wrapper only with actual document.md, page, and DESIGN.md
reads, a concrete no-change report grounded in the fixture's type/palette/layout,
and zero mutations. Seeded files must still be byte-identical and pre-existing
sidecar drift must remain untouched. New-world documentation writes, redesign
ordering, and full-build completion gates are unchanged.
Offline re-evaluation of saved Claude traces: three advice responses and two
evidenced no-op handoffs pass the new assertions. The original post-review
baseline and the 637-second full build still fail for absent documentation
evidence. Unit negative controls reject fabricated prerequisites, unsolicited
edits/interviews/scans, failed reads, empty or unsupported reports, and exhausted
budgets. This is assertion replay, not new model evidence or a rerun of cleaned-up
workspaces' filesystem checks. No paid calls or skill prose changes were needed.
The earlier reference misses above are now diagnostics, not release blockers by
themselves; this does not establish an all-green full-workflow matrix.
Each scenario:
1. `prepareWorkspace()` uses the production transformer to build current source
@@ -241,8 +267,8 @@ results remain the completed measurements.
| 13 | empty workspace; prompt is `/impeccable teach` | runs `impeccable context` and diverts into `reference/init.md` because `teach` aliases `init` |
| 14 | PRODUCT.md with `## Platform: ios` (native iOS app); prompt is `/impeccable craft a tide detail screen` | `impeccable context` runs and emits the contents of `reference/ios.md` directly, placing native conventions in context without a second model-directed read |
| 15 | same iOS fixture; prompt is `/impeccable audit` | agent loads `reference/audit.native.md` (the Commands-table native variant, routed instead of `audit.md`) |
| 16 | existing surface, with and without PRODUCT.md; asks where to start | loads `routing.md`, delivers advice, and does not edit project files, start an interview, archive a critique, or run menu scans |
| 17 | existing surface; asks whether critique is required before polish | loads `routing.md` and both command references, then delivers advice without executing the playbooks |
| 16 | existing surface, with and without PRODUCT.md; asks where to start | completes relevant advice without edits, interviews, critique archives, menu scans, or explicit invented refinement prerequisites; reference coverage is diagnostic |
| 17 | existing surface; asks whether critique is required before polish | completes read-only advice distinguishing assessment from implementation and explaining critique is optional; reference coverage is diagnostic |
| 18 | existing surface; explicitly requests polish followed by a next-command recommendation | loads `polish.md` rather than substituting workflow advice for the requested work |
| 19 | tiny spacing edit with PRODUCT.md + DESIGN.md; Bash denied, a real-loader success control, and a denied-launcher planning-only case | edits require successful playbook/craft-floor reads and a pre-edit denial warning; planning stays read-only and skips craft-floor |
+43
View File
@@ -1,5 +1,48 @@
import assert from 'node:assert/strict';
// These are bounded English-fixture checks, not a general semantic grader.
// Reference coverage is reported separately: opening a file proves neither
// useful advice nor permission to execute it.
export function missingReferences(trace, filenames) {
return filenames.filter((filename) => !trace.toolCalls.some((call) =>
(call.loadedFiles || []).some((file) => file === filename || file.endsWith(`/${filename}`))));
}
export function assertAdviceOnly(trace, text) {
assert.ok(text.trim(), 'advice must reach the user, not stop at reference loading');
assert.deepEqual(trace.writePaths, [], 'advice must not use the write tool');
const mutations = trace.toolCalls.flatMap((call) => call.mutatedPaths ?? [])
.filter((file) => !file.startsWith('.impeccable/') || file.startsWith('.impeccable/critique/'));
assert.deepEqual(mutations, [], 'advice must not edit project files or archive an unsolicited critique');
assert.deepEqual(trace.questionCalls, [], 'advice must not start an init or design interview');
assert.ok(!trace.bashCommands.some((command) => command.includes('impeccable detect')), 'workflow advice does not run menu scans');
}
function normalizedAdvice(text) {
return text.replace(/[`*_]/g, '').replace(/[]/g, "'").toLowerCase();
}
export function assertWorkflowAdvice(trace, text, { missingContext = false } = {}) {
assertAdviceOnly(trace, text);
const advice = normalizedAdvice(text);
assert.match(advice, /index\.html/, 'advice should address the existing surface');
assert.match(advice, missingContext ? /\binit\b/ : /\b(?:critique|audit|polish)\b/, 'advice must recommend a relevant starting point');
if (missingContext) assert.match(advice, /\bdocument\b/, 'advice should explain how to record the existing identity');
assert.doesNotMatch(advice, /(?:must|need to|have to|required to)\s+(?:run\s+)?(?:\/impeccable\s+)?(?:init|document)\b[^.!?\n]{0,100}\bbefore\s+(?:you\s+can\s+)?(?:run(?:ning)?\s+)?(?:polish(?:ing)?|refin(?:e|ing|ement))\b|(?:polish|refinement)\s+(?:requires|is blocked by|cannot run without)\s+(?:init|document|product\.md|design\.md)/,
'setup is not a mandatory prerequisite for narrow refinement');
}
export function assertCommandComparison(trace, text) {
assertAdviceOnly(trace, text);
const advice = normalizedAdvice(text);
assert.match(advice, /critique[^.!?\n]{0,120}(?:review|assess|evaluat|report|findings)/, 'comparison must explain critique as assessment');
assert.match(advice, /polish[^.!?\n]{0,120}(?:fix|refin|implement|edit)/, 'comparison must explain polish as implementation');
assert.match(advice, /critique\s+(?:is\s+)?(?:isn't|is not|not)\s+(?:required|necessary)|critique[^.!?\n]{0,50}\boptional\b|polish[^.!?\n]{0,100}(?:directly|without\s+(?:a\s+)?critique|independent)/,
'comparison must explain that critique is optional before polish');
assert.doesNotMatch(advice, /(?:must|need to|have to)\s+(?:run\s+)?critique[^.!?\n]{0,80}before\s+(?:run(?:ning)?\s+)?polish|critique\s+(?:is\s+)?(?:required|mandatory|necessary)\s+before\s+polish|polish\s+(?:requires|cannot run without)\s+(?:a\s+)?critique/,
'comparison must not invent a critique prerequisite');
}
export function assertNewWorkLifecycle(trace, { target, redesign = false }) {
const calls = trace.toolCalls;
const writes = (call, file) => (call.mutatedPaths || []).includes(file);
+14 -21
View File
@@ -28,7 +28,8 @@ import {
ENGINE_MISSING_MESSAGE,
} from './harness.mjs';
import { detectProvider, getModel, hasKey, resolveModelList, PROVIDERS } from './providers.mjs';
import { assertPlanningFallbackWarning, LAUNCHER_FAILURE_WARNING } from './assertions.mjs';
import { assertPlanningFallbackWarning, LAUNCHER_FAILURE_WARNING, assertAdviceOnly, assertWorkflowAdvice, assertCommandComparison, missingReferences } from './assertions.mjs';
import { assertCompleted } from '../skill-workflow/assertions.mjs';
import {
PRODUCT_MD_SAMPLE,
PRODUCT_MD_SAMPLE_NO_REGISTER,
@@ -90,16 +91,6 @@ function executedUpdateCommands(trace) {
);
}
function assertAdviceOnly(trace, text) {
assert.ok(text.trim(), 'advice must reach the user, not stop at reference loading');
assert.deepEqual(trace.writePaths, [], 'advice must not use the write tool');
const mutations = trace.toolCalls.flatMap((call) => call.mutatedPaths ?? [])
.filter((file) => !file.startsWith('.impeccable/') || file.startsWith('.impeccable/critique/'));
assert.deepEqual(mutations, [], 'advice must not edit project files or archive an unsolicited critique');
assert.deepEqual(trace.questionCalls, [], 'advice must not start an init or design interview');
assert.equal(bashCommandsMatching(trace, 'impeccable detect').length, 0, 'workflow advice does not run menu scans');
}
for (const modelId of resolveModelList()) {
const provider = detectProvider(modelId);
const keyPresent = hasKey(provider);
@@ -634,40 +625,42 @@ for (const modelId of resolveModelList()) {
['existing project', WORKFLOW_ADVICE_FILES],
['missing product context', { 'index.html': MINIMAL_LANDING_HTML }],
]) {
it(`scenario 16: workflow advice stays read-only (${label})`, async () => {
it(`scenario 16: workflow advice stays read-only (${label})`, async (t) => {
const workspace = prepareWorkspace({ files });
try {
const { trace, text } = await runTurn({
const result = await runTurn({
workspace,
model,
userPrompt: "I'm joining this project. Where should I start with Impeccable?",
maxSteps: 8,
contextOnlyBash: true,
});
const { trace, text } = result;
logTrace('S16', label, modelId, trace, { textSample: text.slice(0, 300) });
assert.ok(readsMatching(trace, 'reference/routing.md').length, 'workflow advice loads the shared routing reference');
assertAdviceOnly(trace, text);
t.diagnostic(`Reference coverage gaps (non-blocking): ${missingReferences(trace, ['reference/routing.md']).join(', ') || 'none'}`);
assertCompleted(result);
assertWorkflowAdvice(trace, text, { missingContext: label === 'missing product context' });
} finally {
cleanupWorkspace(workspace);
}
});
}
it('scenario 17: command comparison reads references without running them', async () => {
it('scenario 17: command comparison explains independent commands without running them', async (t) => {
const workspace = prepareWorkspace({ files: WORKFLOW_ADVICE_FILES });
try {
const { trace, text } = await runTurn({
const result = await runTurn({
workspace,
model,
userPrompt: 'Should I use critique or polish on index.html? Is a critique required before polishing?',
maxSteps: 8,
contextOnlyBash: true,
});
const { trace, text } = result;
logTrace('S17', 'command-comparison', modelId, trace, { textSample: text.slice(0, 300) });
assert.ok(readsMatching(trace, 'reference/routing.md').length, 'a command name in a question still routes to advice');
assert.ok(readsMatching(trace, 'reference/critique.md').length, 'comparison consults the critique contract');
assert.ok(readsMatching(trace, 'reference/polish.md').length, 'comparison consults the polish contract');
assertAdviceOnly(trace, text);
t.diagnostic(`Reference coverage gaps (non-blocking): ${missingReferences(trace, ['reference/routing.md', 'reference/critique.md', 'reference/polish.md']).join(', ') || 'none'}`);
assertCompleted(result);
assertCommandComparison(trace, text);
} finally {
cleanupWorkspace(workspace);
}
+20
View File
@@ -1,5 +1,25 @@
import assert from 'node:assert/strict';
import { sourceHash } from './source-hash.mjs';
import { missingReferences } from '../skill-behavior/assertions.mjs';
// For a resumed, already-reviewed ordinary extension only. New worlds and
// redesigns still owe real documentation writes; this is not an escape hatch.
export function assertNoChangeDocumentation(result, { target, evidence }) {
assertCompleted(result);
const { trace, text } = result;
assert.deepEqual(missingReferences(trace, ['reference/document.md', target, 'DESIGN.md']), [],
'documentation must consult its contract and inspect the actual source and recorded system');
assert.deepEqual(trace.toolCalls.flatMap((call) => call.mutatedPaths || []), [],
'the resumed no-change check must not mutate project files');
assert.match(text, /no (?:system |visual.system |documentation )?changes|unchanged|no rewrite/i,
'documentation must explicitly report a no-change outcome');
for (const filename of [target, 'DESIGN.md']) {
assert.ok(text.includes(filename), `documentation must identify the checked ${filename}`);
}
for (const fact of evidence) {
assert.match(text, fact, 'no-change documentation must report evidence from the fixture, not an unsupported completion claim');
}
}
export function assertCompleted(result) {
assert.equal(result.outcome, 'complete', `workflow did not finish: ${result.outcome} after ${result.steps} steps`);
+6 -3
View File
@@ -4,7 +4,8 @@ import fs from 'node:fs';
import path from 'node:path';
import { prepareWorkspace, cleanupWorkspace, runTurn, fileLoaded, ENGINE_BIN } from '../skill-behavior/harness.mjs';
import { getModel, detectProvider, hasKey } from '../skill-behavior/providers.mjs';
import { assertCompleted } from './assertions.mjs';
import { assertCompleted, assertNoChangeDocumentation } from './assertions.mjs';
import { missingReferences } from '../skill-behavior/assertions.mjs';
// A synthetic post-review checkpoint, not another full-build simulation.
// The page and system agree. A missing sidecar predates this task and is not
@@ -29,7 +30,7 @@ const BRIEF = '# Keyboard guide\n\n## Direction contract\nTHESIS: A short readin
for (const modelId of (process.env.IMPECCABLE_SKILL_BEHAVIOR_MODELS || 'claude-sonnet-5').split(',').map((id) => id.trim()).filter(Boolean)) {
for (const existingSystem of [true, false]) {
it(`post-review ${existingSystem ? 'extension preserves' : 'new world records'} its system :: ${modelId}`,
{ skip: !ENGINE_BIN || !hasKey(detectProvider(modelId)) }, async () => {
{ skip: !ENGINE_BIN || !hasKey(detectProvider(modelId)) }, async (t) => {
const files = {
'PRODUCT.md': '# Field Manual\n\n## Platform\nweb\n\nA reference guide for keyboard users.\n',
...(existingSystem ? { 'DESIGN.md': DESIGN } : {}),
@@ -55,7 +56,7 @@ for (const modelId of (process.env.IMPECCABLE_SKILL_BEHAVIOR_MODELS || 'claude-s
userPrompt: 'Continue from this checkpoint and finish the task.',
});
assertCompleted(result);
assert.ok(fileLoaded(result.trace, 'degraded/documenter.md'), 'must load the shipped documentation pass even when DESIGN.md stays unchanged');
t.diagnostic(`Documentation wrapper coverage gaps (non-blocking for evidenced no-op): ${missingReferences(result.trace, ['degraded/documenter.md']).join(', ') || 'none'}`);
assert.ok(fileLoaded(result.trace, 'reference/document.md'), 'must consult the documentation contract');
for (const name of existingSystem ? ['index.html', 'DESIGN.md'] : ['index.html']) {
assert.ok(fileLoaded(result.trace, name), `documentation must check ${name}, not merely announce a no-op`);
@@ -64,9 +65,11 @@ for (const modelId of (process.env.IMPECCABLE_SKILL_BEHAVIOR_MODELS || 'claude-s
assert.equal(fs.readFileSync(path.join(workspace, name), 'utf8'), contents, `${name} must remain unchanged`);
}
if (existingSystem) {
assertNoChangeDocumentation(result, { target: 'index.html', evidence: [/system-ui/i, /65\s*ch/i, /#0645ad/i] });
assert.equal(fs.existsSync(path.join(workspace, '.impeccable/design.json')), false, 'must not repair pre-existing sidecar drift unasked');
assert.deepEqual(result.trace.toolCalls.flatMap((call) => call.mutatedPaths || []), [], 'a no-change check must not mutate other project files');
} else {
assert.ok(fileLoaded(result.trace, 'degraded/documenter.md'), 'new-world documentation must run the shipped documentation pass');
const design = fs.readFileSync(path.join(workspace, 'DESIGN.md'), 'utf8');
assert.match(design, /^---\n/);
assert.match(design, /^colors:/m);