Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 6 additions & 2 deletions scripts/build-pages-report.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -421,6 +421,7 @@ function validReportEntry(entry, files) {
testCase.selection === 'workflow-failure') &&
[
'last-screenshot',
'last-no-screenshot',
'first-failing-screenshot',
'first-failing-no-screenshot',
'workflow-failure',
Expand All @@ -432,7 +433,9 @@ function validReportEntry(entry, files) {
['ai', 'error', 'result'].includes(testCase.descriptionKind)) &&
(testCase.description === undefined) ===
(testCase.descriptionKind === undefined) &&
(testCase.selection === 'first-failing-no-screenshot'
(['first-failing-no-screenshot', 'last-no-screenshot'].includes(
testCase.selection,
)
? testCase.previewPath === undefined
: files?.includes(testCase.previewPath)) &&
(testCase.reportPath === undefined ||
Expand Down Expand Up @@ -1058,7 +1061,8 @@ export async function buildPagesReport(options) {
for (const testCase of entry.cases) {
if (
testCase.selection === 'workflow-failure' ||
testCase.selection === 'first-failing-no-screenshot'
testCase.selection === 'first-failing-no-screenshot' ||
testCase.selection === 'last-no-screenshot'
) {
continue;
}
Expand Down
162 changes: 98 additions & 64 deletions scripts/report-cases.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -114,10 +114,9 @@ function executionForDetail(dumps, detail) {
return null;
}

function evidenceForStep(step, embedded) {
function evidenceForStep(step, embedded, allowUntimedEvidence = false) {
const candidates = [];
for (const detail of step.agentDetails ?? []) {
const execution = executionForDetail(embedded.dumps, detail);
function collect(execution, destination) {
for (const task of execution?.tasks ?? []) {
const screenshotId = task?.uiContext?.screenshot?.id;
const explanation = modelTaskText(task);
Expand All @@ -126,17 +125,42 @@ function evidenceForStep(step, embedded) {
screenshotId &&
embedded.images.has(screenshotId)
) {
candidates.push({ screenshotId, explanation });
destination.push({ screenshotId, explanation });
}
}
}
for (const detail of step.agentDetails ?? []) {
collect(executionForDetail(embedded.dumps, detail), candidates);
}
// Custom Nodes can call an Agent without forwarding agentDetails to the
// runner step. Match by execution time when available; for older reports
// without timestamps, only a one-step, one-attempt case is unambiguous.
if (!candidates.length && !hasScreenshotEvidence(step)) {
const startedAt = Date.parse(step.startedAt ?? '');
const endedAt = Date.parse(step.endedAt ?? '');
const hasWindow = Number.isFinite(startedAt) && Number.isFinite(endedAt);
const unlinked = [];
for (const entry of embedded.dumps) {
for (const execution of entry.dump?.executions ?? []) {
const executionTime = Number(
execution.logTime ?? execution.tasks?.[0]?.timing?.start,
);
const inStep = hasWindow && Number.isFinite(executionTime) &&
executionTime >= startedAt - 1000 &&
executionTime <= endedAt + 1000;
if (!inStep && !(allowUntimedEvidence && !hasWindow)) continue;
collect(execution, unlinked);
}
}
if (unlinked.length === 1) candidates.push(unlinked[0]);
}
const selected = candidates.at(-1);
const error = normalizedText(step.error?.message);
const result = normalizedText(step.output?.summary);
const description = error ?? selected?.explanation ?? result;
if (!description) {
throw new Error(`Step ${step.id} has no AI response or error text`);
}
const description = error ?? selected?.explanation ?? result ??
(step.status === 'success'
? 'Step passed; no per-step AI response was recorded.'
: 'Step failed without a recorded error message.');
return {
...(selected
? { screenshot: embedded.images.get(selected.screenshotId) }
Expand All @@ -159,61 +183,71 @@ export async function reportCases(
const embedded = reportHtml
? await embeddedReportData(reportHtml, reportFile)
: null;
return (project.documents ?? []).flatMap((document) =>
(document.cases ?? []).flatMap((testCase) => {
const attempt = testCase.attempts?.at(-1);
if (!attempt) {
if (testCase.status === 'not-run') return [];
throw new Error(
`Case ${testCase.name ?? testCase.caseId} has no attempt`,
);
}
const steps = allAttemptSteps(attempt);
const passed = (testCase.status ?? attempt.status) === 'success';
const step = passed
? steps.findLast(hasScreenshotEvidence) ?? steps.at(-1)
: steps.find(
(item) => item.status === 'failed' && hasScreenshotEvidence(item),
) ??
steps.find((item) => item.status === 'failed');
if (!step?.id) {
throw new Error(
`Case ${testCase.name ?? testCase.caseId} has no report step to preview`,
);
}
if (!testCase.caseId || !testCase.name) {
throw new Error('Midscene case metadata is incomplete');
}
const evidence = embedded ? evidenceForStep(step, embedded) : null;
if (passed && embedded && !evidence?.screenshot) {
throw new Error(`Step ${step.id} has no embedded node screenshot`);
}
const selection = passed
? 'last-screenshot'
: evidence && !evidence.screenshot
? 'first-failing-no-screenshot'
: 'first-failing-screenshot';
return [{
caseId: testCase.caseId,
name: testCase.name,
status: passed ? 'success' : 'failed',
durationMs: attempt.durationMs,
stepId: step.id,
stepTitle: step.title ?? step.node,
selection,
...(evidence?.screenshot
? {
previewFile: casePreviewFileName(
projectName,
testCase.caseId,
evidence.screenshot.extension,
),
}
: embedded
? {}
: { previewFile: casePreviewFileName(projectName, testCase.caseId) }),
...(evidence ?? {}),
}];
}),
const cases = (project.documents ?? []).flatMap((document) =>
document.cases ?? [],
);
return cases.flatMap((testCase) => {
const attempt = testCase.attempts?.at(-1);
if (!attempt) {
if (testCase.status === 'not-run') return [];
throw new Error(
`Case ${testCase.name ?? testCase.caseId} has no attempt`,
);
}
const steps = allAttemptSteps(attempt);
const passed = (testCase.status ?? attempt.status) === 'success';
const step = passed
? steps.findLast(hasScreenshotEvidence) ?? steps.at(-1)
: steps.find(
(item) => item.status === 'failed' && hasScreenshotEvidence(item),
) ??
steps.find((item) => item.status === 'failed');
if (!step?.id) {
throw new Error(
`Case ${testCase.name ?? testCase.caseId} has no report step to preview`,
);
}
if (!testCase.caseId || !testCase.name) {
throw new Error('Midscene case metadata is incomplete');
}
const soleUntimedStep = cases.length === 1 &&
testCase.attempts?.length === 1 && steps.length === 1;
const evidence = embedded
? evidenceForStep(step, embedded, soleUntimedStep)
: null;
if (
passed && embedded && !evidence?.screenshot &&
(hasScreenshotEvidence(step) || step.node?.startsWith('ai'))
) {
throw new Error(`Step ${step.id} has no embedded node screenshot`);
}
const selection = passed
? embedded && !evidence?.screenshot
? 'last-no-screenshot'
: 'last-screenshot'
: evidence && !evidence.screenshot
? 'first-failing-no-screenshot'
: 'first-failing-screenshot';
return [{
caseId: testCase.caseId,
name: testCase.name,
status: passed ? 'success' : 'failed',
durationMs: attempt.durationMs,
stepId: step.id,
stepTitle: step.title ?? step.node,
selection,
...(evidence?.screenshot
? {
previewFile: casePreviewFileName(
projectName,
testCase.caseId,
evidence.screenshot.extension,
),
}
: embedded
? {}
: { previewFile: casePreviewFileName(projectName, testCase.caseId) }),
...(evidence ?? {}),
}];
});
}
100 changes: 100 additions & 0 deletions tests/contracts/test_pages_report_history.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -1667,6 +1667,106 @@ test('pairs an original node screenshot with its AI text or error', async () =>
assert.deepEqual(failure.screenshot.bytes, Buffer.from('/9j/2Q==', 'base64'));
});

test('recovers a sole Agent screenshot from an unlinked custom visual step', async () => {
const html = runnerScript({
project: 'ubuntu',
startedAt: '2026-09-26T07:32:45Z',
});
const run = testRunDump(html);
const step = run.projects[0].documents[0].cases[0].attempts[0].steps[0];
step.node = 'review.assertConfiguredVisual';
delete step.agentDetails;

const [testCase] = await reportCases(run, 'ubuntu', { reportHtml: html });
assert.equal(testCase.selection, 'last-screenshot');
assert.equal(testCase.descriptionKind, 'ai');
assert.equal(testCase.description, 'AI explanation 0-0');
assert.deepEqual(testCase.screenshot.bytes, Buffer.from('/9j/2Q==', 'base64'));
});

test('matches unlinked custom-step screenshots by execution time', async () => {
const html = runnerScript({
assertionCount: 2,
project: 'ubuntu',
startedAt: '2026-09-26T07:32:45Z',
});
const run = testRunDump(html);
const steps = run.projects[0].documents[0].cases[0].attempts[0].steps;
for (const [index, step] of steps.entries()) {
step.node = 'review.assertConfiguredVisual';
step.startedAt = `2026-09-26T07:32:${index ? '55' : '45'}.000Z`;
step.endedAt = `2026-09-26T07:33:${index ? '05' : '00'}.000Z`;
delete step.agentDetails;
}
const dumpMatch = html.match(/<script type="midscene_web_dump"[^>]*>(\{[\s\S]*?)<\/script>/);
const dump = JSON.parse(dumpMatch[1]);
dump.executions[0].logTime = Date.parse('2026-09-26T07:32:47.000Z');
dump.executions[1].logTime = Date.parse('2026-09-26T07:32:57.000Z');
const reportHtml = html.replace(dumpMatch[1], JSON.stringify(dump));

const [testCase] = await reportCases(run, 'ubuntu', { reportHtml });
assert.equal(testCase.stepId, steps[1].id);
assert.equal(testCase.selection, 'last-screenshot');
assert.equal(testCase.description, 'AI explanation 0-1');
});

test('keeps an ambiguous custom-step success in the table without inventing a screenshot', async (context) => {
const html = runnerScript({
assertionCount: 2,
project: 'ubuntu',
startedAt: '2026-09-26T07:32:45Z',
});
const run = testRunDump(html);
for (const step of run.projects[0].documents[0].cases[0].attempts[0].steps) {
step.node = 'review.assertConfiguredVisual';
delete step.agentDetails;
}
const reportHtml = html.replace(
/<script type="midscene_test_run_dump">[\s\S]*?<\/script>/,
`<script type="midscene_test_run_dump">${JSON.stringify(run)}</script>`,
);
const [testCase] = await reportCases(run, 'ubuntu', { reportHtml });
assert.equal(testCase.selection, 'last-no-screenshot');
assert.equal(testCase.screenshot, undefined);
assert.equal(testCase.descriptionKind, 'result');

const root = await mkdtemp(path.join(os.tmpdir(), 'pages-unlinked-step-'));
const reportDirectory = path.join(root, 'artifact');
await mkdir(path.join(reportDirectory, 'report'), { recursive: true });
await writeFile(path.join(reportDirectory, 'report', 'test-run-ubuntu.html'), reportHtml);
await writeFile(path.join(reportDirectory, 'report-preview.png'), 'preview');
const server = await startServer((_request, response) => response.writeHead(404).end());
context.after(server.close);
const manifest = await buildPagesReport(
options(reportDirectory, path.join(root, 'site'), server.url),
);
const [publishedCase] = manifest.reports[0].entries[0].cases;
assert.equal(publishedCase.status, 'success');
assert.equal(publishedCase.selection, 'last-no-screenshot');
assert.equal(publishedCase.previewPath, undefined);
assert.equal(publishedCase.description, 'Step passed; no per-step AI response was recorded.');
const summary = renderReportSummary({
manifest,
pagesUrl: server.url,
producerResult: 'success',
runId: '200',
summaryTitle: 'Ubuntu',
});
assert.match(summary, /Passed case|ubuntu visual case/);
assert.match(summary, /\| — \| ✅ Passed \|/);
});

test('still rejects missing screenshots from explicitly linked AI steps', async () => {
const html = runnerScript({
project: 'ubuntu',
startedAt: '2026-09-26T07:32:45Z',
}).replace(/<script type="midscene-image"[^>]*>[\s\S]*?<\/script>/g, '');
await assert.rejects(
reportCases(testRunDump(html), 'ubuntu', { reportHtml: html }),
/has no embedded node screenshot/,
);
});

test('preserves a failed case and skips a not-run case after agent damage', async () => {
const html = runnerWithoutScreenshot({
project: 'ubuntu',
Expand Down
Loading