Skip to content

Commit 6426384

Browse files
authored
Merge pull request #59 from quanru/fix/report-custom-step-evidence
fix: keep Midscene CI report stable for custom visual steps
2 parents db5fc28 + 9bb12dc commit 6426384

3 files changed

Lines changed: 204 additions & 66 deletions

File tree

‎scripts/build-pages-report.mjs‎

Lines changed: 6 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -421,6 +421,7 @@ function validReportEntry(entry, files) {
421421
testCase.selection === 'workflow-failure') &&
422422
[
423423
'last-screenshot',
424+
'last-no-screenshot',
424425
'first-failing-screenshot',
425426
'first-failing-no-screenshot',
426427
'workflow-failure',
@@ -432,7 +433,9 @@ function validReportEntry(entry, files) {
432433
['ai', 'error', 'result'].includes(testCase.descriptionKind)) &&
433434
(testCase.description === undefined) ===
434435
(testCase.descriptionKind === undefined) &&
435-
(testCase.selection === 'first-failing-no-screenshot'
436+
(['first-failing-no-screenshot', 'last-no-screenshot'].includes(
437+
testCase.selection,
438+
)
436439
? testCase.previewPath === undefined
437440
: files?.includes(testCase.previewPath)) &&
438441
(testCase.reportPath === undefined ||
@@ -1058,7 +1061,8 @@ export async function buildPagesReport(options) {
10581061
for (const testCase of entry.cases) {
10591062
if (
10601063
testCase.selection === 'workflow-failure' ||
1061-
testCase.selection === 'first-failing-no-screenshot'
1064+
testCase.selection === 'first-failing-no-screenshot' ||
1065+
testCase.selection === 'last-no-screenshot'
10621066
) {
10631067
continue;
10641068
}

‎scripts/report-cases.mjs‎

Lines changed: 98 additions & 64 deletions
Original file line numberDiff line numberDiff line change
@@ -114,10 +114,9 @@ function executionForDetail(dumps, detail) {
114114
return null;
115115
}
116116

117-
function evidenceForStep(step, embedded) {
117+
function evidenceForStep(step, embedded, allowUntimedEvidence = false) {
118118
const candidates = [];
119-
for (const detail of step.agentDetails ?? []) {
120-
const execution = executionForDetail(embedded.dumps, detail);
119+
function collect(execution, destination) {
121120
for (const task of execution?.tasks ?? []) {
122121
const screenshotId = task?.uiContext?.screenshot?.id;
123122
const explanation = modelTaskText(task);
@@ -126,17 +125,42 @@ function evidenceForStep(step, embedded) {
126125
screenshotId &&
127126
embedded.images.has(screenshotId)
128127
) {
129-
candidates.push({ screenshotId, explanation });
128+
destination.push({ screenshotId, explanation });
130129
}
131130
}
132131
}
132+
for (const detail of step.agentDetails ?? []) {
133+
collect(executionForDetail(embedded.dumps, detail), candidates);
134+
}
135+
// Custom Nodes can call an Agent without forwarding agentDetails to the
136+
// runner step. Match by execution time when available; for older reports
137+
// without timestamps, only a one-step, one-attempt case is unambiguous.
138+
if (!candidates.length && !hasScreenshotEvidence(step)) {
139+
const startedAt = Date.parse(step.startedAt ?? '');
140+
const endedAt = Date.parse(step.endedAt ?? '');
141+
const hasWindow = Number.isFinite(startedAt) && Number.isFinite(endedAt);
142+
const unlinked = [];
143+
for (const entry of embedded.dumps) {
144+
for (const execution of entry.dump?.executions ?? []) {
145+
const executionTime = Number(
146+
execution.logTime ?? execution.tasks?.[0]?.timing?.start,
147+
);
148+
const inStep = hasWindow && Number.isFinite(executionTime) &&
149+
executionTime >= startedAt - 1000 &&
150+
executionTime <= endedAt + 1000;
151+
if (!inStep && !(allowUntimedEvidence && !hasWindow)) continue;
152+
collect(execution, unlinked);
153+
}
154+
}
155+
if (unlinked.length === 1) candidates.push(unlinked[0]);
156+
}
133157
const selected = candidates.at(-1);
134158
const error = normalizedText(step.error?.message);
135159
const result = normalizedText(step.output?.summary);
136-
const description = error ?? selected?.explanation ?? result;
137-
if (!description) {
138-
throw new Error(`Step ${step.id} has no AI response or error text`);
139-
}
160+
const description = error ?? selected?.explanation ?? result ??
161+
(step.status === 'success'
162+
? 'Step passed; no per-step AI response was recorded.'
163+
: 'Step failed without a recorded error message.');
140164
return {
141165
...(selected
142166
? { screenshot: embedded.images.get(selected.screenshotId) }
@@ -159,61 +183,71 @@ export async function reportCases(
159183
const embedded = reportHtml
160184
? await embeddedReportData(reportHtml, reportFile)
161185
: null;
162-
return (project.documents ?? []).flatMap((document) =>
163-
(document.cases ?? []).flatMap((testCase) => {
164-
const attempt = testCase.attempts?.at(-1);
165-
if (!attempt) {
166-
if (testCase.status === 'not-run') return [];
167-
throw new Error(
168-
`Case ${testCase.name ?? testCase.caseId} has no attempt`,
169-
);
170-
}
171-
const steps = allAttemptSteps(attempt);
172-
const passed = (testCase.status ?? attempt.status) === 'success';
173-
const step = passed
174-
? steps.findLast(hasScreenshotEvidence) ?? steps.at(-1)
175-
: steps.find(
176-
(item) => item.status === 'failed' && hasScreenshotEvidence(item),
177-
) ??
178-
steps.find((item) => item.status === 'failed');
179-
if (!step?.id) {
180-
throw new Error(
181-
`Case ${testCase.name ?? testCase.caseId} has no report step to preview`,
182-
);
183-
}
184-
if (!testCase.caseId || !testCase.name) {
185-
throw new Error('Midscene case metadata is incomplete');
186-
}
187-
const evidence = embedded ? evidenceForStep(step, embedded) : null;
188-
if (passed && embedded && !evidence?.screenshot) {
189-
throw new Error(`Step ${step.id} has no embedded node screenshot`);
190-
}
191-
const selection = passed
192-
? 'last-screenshot'
193-
: evidence && !evidence.screenshot
194-
? 'first-failing-no-screenshot'
195-
: 'first-failing-screenshot';
196-
return [{
197-
caseId: testCase.caseId,
198-
name: testCase.name,
199-
status: passed ? 'success' : 'failed',
200-
durationMs: attempt.durationMs,
201-
stepId: step.id,
202-
stepTitle: step.title ?? step.node,
203-
selection,
204-
...(evidence?.screenshot
205-
? {
206-
previewFile: casePreviewFileName(
207-
projectName,
208-
testCase.caseId,
209-
evidence.screenshot.extension,
210-
),
211-
}
212-
: embedded
213-
? {}
214-
: { previewFile: casePreviewFileName(projectName, testCase.caseId) }),
215-
...(evidence ?? {}),
216-
}];
217-
}),
186+
const cases = (project.documents ?? []).flatMap((document) =>
187+
document.cases ?? [],
218188
);
189+
return cases.flatMap((testCase) => {
190+
const attempt = testCase.attempts?.at(-1);
191+
if (!attempt) {
192+
if (testCase.status === 'not-run') return [];
193+
throw new Error(
194+
`Case ${testCase.name ?? testCase.caseId} has no attempt`,
195+
);
196+
}
197+
const steps = allAttemptSteps(attempt);
198+
const passed = (testCase.status ?? attempt.status) === 'success';
199+
const step = passed
200+
? steps.findLast(hasScreenshotEvidence) ?? steps.at(-1)
201+
: steps.find(
202+
(item) => item.status === 'failed' && hasScreenshotEvidence(item),
203+
) ??
204+
steps.find((item) => item.status === 'failed');
205+
if (!step?.id) {
206+
throw new Error(
207+
`Case ${testCase.name ?? testCase.caseId} has no report step to preview`,
208+
);
209+
}
210+
if (!testCase.caseId || !testCase.name) {
211+
throw new Error('Midscene case metadata is incomplete');
212+
}
213+
const soleUntimedStep = cases.length === 1 &&
214+
testCase.attempts?.length === 1 && steps.length === 1;
215+
const evidence = embedded
216+
? evidenceForStep(step, embedded, soleUntimedStep)
217+
: null;
218+
if (
219+
passed && embedded && !evidence?.screenshot &&
220+
(hasScreenshotEvidence(step) || step.node?.startsWith('ai'))
221+
) {
222+
throw new Error(`Step ${step.id} has no embedded node screenshot`);
223+
}
224+
const selection = passed
225+
? embedded && !evidence?.screenshot
226+
? 'last-no-screenshot'
227+
: 'last-screenshot'
228+
: evidence && !evidence.screenshot
229+
? 'first-failing-no-screenshot'
230+
: 'first-failing-screenshot';
231+
return [{
232+
caseId: testCase.caseId,
233+
name: testCase.name,
234+
status: passed ? 'success' : 'failed',
235+
durationMs: attempt.durationMs,
236+
stepId: step.id,
237+
stepTitle: step.title ?? step.node,
238+
selection,
239+
...(evidence?.screenshot
240+
? {
241+
previewFile: casePreviewFileName(
242+
projectName,
243+
testCase.caseId,
244+
evidence.screenshot.extension,
245+
),
246+
}
247+
: embedded
248+
? {}
249+
: { previewFile: casePreviewFileName(projectName, testCase.caseId) }),
250+
...(evidence ?? {}),
251+
}];
252+
});
219253
}

‎tests/contracts/test_pages_report_history.mjs‎

Lines changed: 100 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1667,6 +1667,106 @@ test('pairs an original node screenshot with its AI text or error', async () =>
16671667
assert.deepEqual(failure.screenshot.bytes, Buffer.from('/9j/2Q==', 'base64'));
16681668
});
16691669

1670+
test('recovers a sole Agent screenshot from an unlinked custom visual step', async () => {
1671+
const html = runnerScript({
1672+
project: 'ubuntu',
1673+
startedAt: '2026-09-26T07:32:45Z',
1674+
});
1675+
const run = testRunDump(html);
1676+
const step = run.projects[0].documents[0].cases[0].attempts[0].steps[0];
1677+
step.node = 'review.assertConfiguredVisual';
1678+
delete step.agentDetails;
1679+
1680+
const [testCase] = await reportCases(run, 'ubuntu', { reportHtml: html });
1681+
assert.equal(testCase.selection, 'last-screenshot');
1682+
assert.equal(testCase.descriptionKind, 'ai');
1683+
assert.equal(testCase.description, 'AI explanation 0-0');
1684+
assert.deepEqual(testCase.screenshot.bytes, Buffer.from('/9j/2Q==', 'base64'));
1685+
});
1686+
1687+
test('matches unlinked custom-step screenshots by execution time', async () => {
1688+
const html = runnerScript({
1689+
assertionCount: 2,
1690+
project: 'ubuntu',
1691+
startedAt: '2026-09-26T07:32:45Z',
1692+
});
1693+
const run = testRunDump(html);
1694+
const steps = run.projects[0].documents[0].cases[0].attempts[0].steps;
1695+
for (const [index, step] of steps.entries()) {
1696+
step.node = 'review.assertConfiguredVisual';
1697+
step.startedAt = `2026-09-26T07:32:${index ? '55' : '45'}.000Z`;
1698+
step.endedAt = `2026-09-26T07:33:${index ? '05' : '00'}.000Z`;
1699+
delete step.agentDetails;
1700+
}
1701+
const dumpMatch = html.match(/<script type="midscene_web_dump"[^>]*>(\{[\s\S]*?)<\/script>/);
1702+
const dump = JSON.parse(dumpMatch[1]);
1703+
dump.executions[0].logTime = Date.parse('2026-09-26T07:32:47.000Z');
1704+
dump.executions[1].logTime = Date.parse('2026-09-26T07:32:57.000Z');
1705+
const reportHtml = html.replace(dumpMatch[1], JSON.stringify(dump));
1706+
1707+
const [testCase] = await reportCases(run, 'ubuntu', { reportHtml });
1708+
assert.equal(testCase.stepId, steps[1].id);
1709+
assert.equal(testCase.selection, 'last-screenshot');
1710+
assert.equal(testCase.description, 'AI explanation 0-1');
1711+
});
1712+
1713+
test('keeps an ambiguous custom-step success in the table without inventing a screenshot', async (context) => {
1714+
const html = runnerScript({
1715+
assertionCount: 2,
1716+
project: 'ubuntu',
1717+
startedAt: '2026-09-26T07:32:45Z',
1718+
});
1719+
const run = testRunDump(html);
1720+
for (const step of run.projects[0].documents[0].cases[0].attempts[0].steps) {
1721+
step.node = 'review.assertConfiguredVisual';
1722+
delete step.agentDetails;
1723+
}
1724+
const reportHtml = html.replace(
1725+
/<script type="midscene_test_run_dump">[\s\S]*?<\/script>/,
1726+
`<script type="midscene_test_run_dump">${JSON.stringify(run)}</script>`,
1727+
);
1728+
const [testCase] = await reportCases(run, 'ubuntu', { reportHtml });
1729+
assert.equal(testCase.selection, 'last-no-screenshot');
1730+
assert.equal(testCase.screenshot, undefined);
1731+
assert.equal(testCase.descriptionKind, 'result');
1732+
1733+
const root = await mkdtemp(path.join(os.tmpdir(), 'pages-unlinked-step-'));
1734+
const reportDirectory = path.join(root, 'artifact');
1735+
await mkdir(path.join(reportDirectory, 'report'), { recursive: true });
1736+
await writeFile(path.join(reportDirectory, 'report', 'test-run-ubuntu.html'), reportHtml);
1737+
await writeFile(path.join(reportDirectory, 'report-preview.png'), 'preview');
1738+
const server = await startServer((_request, response) => response.writeHead(404).end());
1739+
context.after(server.close);
1740+
const manifest = await buildPagesReport(
1741+
options(reportDirectory, path.join(root, 'site'), server.url),
1742+
);
1743+
const [publishedCase] = manifest.reports[0].entries[0].cases;
1744+
assert.equal(publishedCase.status, 'success');
1745+
assert.equal(publishedCase.selection, 'last-no-screenshot');
1746+
assert.equal(publishedCase.previewPath, undefined);
1747+
assert.equal(publishedCase.description, 'Step passed; no per-step AI response was recorded.');
1748+
const summary = renderReportSummary({
1749+
manifest,
1750+
pagesUrl: server.url,
1751+
producerResult: 'success',
1752+
runId: '200',
1753+
summaryTitle: 'Ubuntu',
1754+
});
1755+
assert.match(summary, /Passed case|ubuntu visual case/);
1756+
assert.match(summary, /\| — \| ✅ Passed \|/);
1757+
});
1758+
1759+
test('still rejects missing screenshots from explicitly linked AI steps', async () => {
1760+
const html = runnerScript({
1761+
project: 'ubuntu',
1762+
startedAt: '2026-09-26T07:32:45Z',
1763+
}).replace(/<script type="midscene-image"[^>]*>[\s\S]*?<\/script>/g, '');
1764+
await assert.rejects(
1765+
reportCases(testRunDump(html), 'ubuntu', { reportHtml: html }),
1766+
/has no embedded node screenshot/,
1767+
);
1768+
});
1769+
16701770
test('preserves a failed case and skips a not-run case after agent damage', async () => {
16711771
const html = runnerWithoutScreenshot({
16721772
project: 'ubuntu',

0 commit comments

Comments
 (0)