') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); })(); [ci] Retry e2e tests once in CI, keeping retried tests visible by alangenfeld · Pull Request #3530 · vercel/workflow · GitHub
Skip to content

[ci] Retry e2e tests once in CI, keeping retried tests visible - #3530

Merged
alangenfeld merged 2 commits into
mainfrom
alangenfeld/e2e-ci-retry
Aug 14, 2026
Merged

[ci] Retry e2e tests once in CI, keeping retried tests visible#3530
alangenfeld merged 2 commits into
mainfrom
alangenfeld/e2e-ci-retry

Conversation

@alangenfeld

Copy link
Copy Markdown
Collaborator

Summary & Motivation

Over the last 10 days ~93 Tests runs needed a human to click re-run until green (some took 6 attempts), because one test losing a timing race fails a whole 20+ minute e2e matrix job.

  • retry: 1 in CI only, so races still reproduce locally while debugging.
  • Retried-then-passed tests stay visible everywhere a failure would have been: ::warning annotations, an e2e-flaky-*.json sidecar per job, and a flaky section in the step summary and PR comment with per-app occurrence counts.
  • event-log-race-repro and the benchmarks pin retry: 0 — their failures are the measurement.

Test Plan

Reporter smoke-tested against a synthetic retried test and the aggregation script against synthetic artifact directories in both modes; confirmed the two retry: 0 suites still collect.

Over the last 10 days ~93 Tests runs were manually re-run until green,
some taking 6 attempts: the e2e suites drive real deployments, and a
single test losing a timing race fails a whole 20+ minute matrix job.
A CI-only vitest retry (retry: 1) absorbs those single-test races.
beforeEach/afterEach hooks run per attempt, so suites with file-restore
hooks (dev.test.ts) retry cleanly.
Retries must not hide real races, so a retried-then-passed test stays
visible everywhere a failure would have been: the github-reporter emits
::warning annotations and an e2e-flaky-*.json sidecar, every e2e job
uploads it, and aggregate-e2e-results.js renders a 'Flaky E2E Tests
(passed on retry)' section in both the per-job step summary and the PR
comment, with per-app occurrence counts.
Harnesses whose failures are themselves the signal pin retry: 0:
event-log-race-repro (a pass runs the full configured budget) and
benchmarks (a regression should not be papered over by a luckier
second sample). Local runs keep retry at 0 so races reproduce while
debugging.
Signed-off-by: Alex Langenfeld <alex.langenfeld@vercel.com>
Signed-off-by: Alex Langenfeld <alex.langenfeld@vercel.com>
@changeset-bot

Copy link
Copy Markdown

🦋 Changeset detected

Latest commit: b40e1aa

The changes in this PR will be included in the next version bump.

This PR includes changesets to release 16 packages
NameType
@workflow/corePatch
@workflow/buildersPatch
@workflow/cliPatch
@workflow/nextPatch
@workflow/nitroPatch
@workflow/vitestPatch
@workflow/web-sharedPatch
@workflow/webPatch
workflowPatch
@workflow/world-testingPatch
@workflow/astroPatch
@workflow/nestPatch
@workflow/rollupPatch
@workflow/sveltekitPatch
@workflow/vitePatch
@workflow/nuxtPatch

Not sure what this means? Click here to learn what changesets are.

Click here if you're a maintainer who wants to add another changeset to this PR

@vercel

vercelBot commented Aug 13, 2026

Copy link
Copy Markdown
Contributor

The latest updates on your projects. Learn more about Vercel for GitHub.

ProjectDeploymentActionsUpdated (UTC)
example-nextjs-workflow-turbopackReadyReadyPreviewAug 13, 2026 10:30pm
example-nextjs-workflow-webpackReadyReadyPreviewAug 13, 2026 10:30pm
example-workflowReadyReadyPreviewAug 13, 2026 10:30pm
workbench-astro-workflowReadyReadyPreviewAug 13, 2026 10:30pm
workbench-express-workflowReadyReadyPreviewAug 13, 2026 10:30pm
workbench-fastify-workflowReadyReadyPreviewAug 13, 2026 10:30pm
workbench-hono-workflowReadyReadyPreviewAug 13, 2026 10:30pm
workbench-nestjs-workflowReadyReadyPreviewAug 13, 2026 10:30pm
workbench-nitro-workflowReadyReadyPreviewAug 13, 2026 10:30pm
workbench-nuxt-workflowReadyReadyPreviewAug 13, 2026 10:30pm
workbench-python-workflowErrorErrorAug 13, 2026 10:30pm
workbench-sveltekit-workflowReadyReadyPreviewAug 13, 2026 10:30pm
workbench-tanstack-start-workflowReadyReadyPreviewAug 13, 2026 10:30pm
workbench-vite-workflowReadyReadyPreviewAug 13, 2026 10:30pm
workflow-docsReadyReadyPreview, v0Aug 13, 2026 10:30pm
workflow-swc-playgroundReadyReadyPreviewAug 13, 2026 10:30pm
workflow-tarballsReadyReadyPreviewAug 13, 2026 10:30pm
workflow-webReadyReadyPreviewAug 13, 2026 10:30pm

@github-actions

github-actionsBot commented Aug 13, 2026

Copy link
Copy Markdown
Contributor

🧪 E2E Test Results

All tests passed

⚠️ Flaky E2E Tests (passed on retry)

These tests failed at least once and passed on a retry. A recurring entry here is a real race worth investigating.

  • concurrent hook token conflict - two workflows cannot use the same hook token simultaneously (vite)
  • hookCleanupTestWorkflow - hook token reuse after workflow completion (astro)
  • hookCleanupTestWorkflow - hook token reuse after workflow completion (sveltekit)
  • step throw of a non-Error value preserves it as cause on the wrapping FatalError (hono)

E2E Test Summary

Summary
PassedFailedSkippedTotal
✅ ▲ Vercel Production346605904056
✅ 💻 Local Development381005584368
✅ 📦 Local Production381005584368
✅ 🐘 Local Postgres381005584368
✅ 🪟 Windows31200312
✅ vercel-multi-region270027
Total152350226417499
Details by Category

✅ ▲ Vercel Production

AppPassedFailedSkipped
✅ astro-node128028
✅ astro-quickjs128028
✅ example-node128028
✅ example-quickjs128028
✅ express-node128028
✅ express-quickjs128028
✅ fastify-node128028
✅ fastify-quickjs128028
✅ hono-node128028
✅ hono-quickjs128028
✅ nest-node128028
✅ nest-quickjs128028
✅ nextjs-turbopack-node15303
✅ nextjs-turbopack-quickjs15303
✅ nextjs-webpack-node15303
✅ nextjs-webpack-quickjs15303
✅ nitro-node128028
✅ nitro-quickjs128028
✅ nuxt-node128028
✅ nuxt-quickjs128028
✅ sveltekit-node14709
✅ sveltekit-quickjs14709
✅ tanstack-start-node128028
✅ tanstack-start-quickjs128028
✅ vite-node128028
✅ vite-quickjs128028

✅ 💻 Local Development

AppPassedFailedSkipped
✅ astro-stable-node130026
✅ astro-stable-quickjs130026
✅ express-stable-node130026
✅ express-stable-quickjs130026
✅ fastify-stable-node130026
✅ fastify-stable-quickjs130026
✅ hono-stable-node130026
✅ hono-stable-quickjs130026
✅ nest-stable-node130026
✅ nest-stable-quickjs130026
✅ nextjs-turbopack-canary-node137019
✅ nextjs-turbopack-canary-quickjs137019
✅ nextjs-turbopack-stable-node15600
✅ nextjs-turbopack-stable-quickjs15600
✅ nextjs-webpack-canary-node137019
✅ nextjs-webpack-canary-quickjs137019
✅ nextjs-webpack-stable-node15600
✅ nextjs-webpack-stable-quickjs15600
✅ nitro-stable-node130026
✅ nitro-stable-quickjs130026
✅ nuxt-stable-node130026
✅ nuxt-stable-quickjs130026
✅ sveltekit-stable-node14907
✅ sveltekit-stable-quickjs14907
✅ tanstack-start-node130026
✅ tanstack-start-quickjs130026
✅ vite-stable-node130026
✅ vite-stable-quickjs130026

✅ 📦 Local Production

AppPassedFailedSkipped
✅ astro-stable-node130026
✅ astro-stable-quickjs130026
✅ express-stable-node130026
✅ express-stable-quickjs130026
✅ fastify-stable-node130026
✅ fastify-stable-quickjs130026
✅ hono-stable-node130026
✅ hono-stable-quickjs130026
✅ nest-stable-node130026
✅ nest-stable-quickjs130026
✅ nextjs-turbopack-canary-node137019
✅ nextjs-turbopack-canary-quickjs137019
✅ nextjs-turbopack-stable-node15600
✅ nextjs-turbopack-stable-quickjs15600
✅ nextjs-webpack-canary-node137019
✅ nextjs-webpack-canary-quickjs137019
✅ nextjs-webpack-stable-node15600
✅ nextjs-webpack-stable-quickjs15600
✅ nitro-stable-node130026
✅ nitro-stable-quickjs130026
✅ nuxt-stable-node130026
✅ nuxt-stable-quickjs130026
✅ sveltekit-stable-node14907
✅ sveltekit-stable-quickjs14907
✅ tanstack-start-node130026
✅ tanstack-start-quickjs130026
✅ vite-stable-node130026
✅ vite-stable-quickjs130026

✅ 🐘 Local Postgres

AppPassedFailedSkipped
✅ astro-stable-node130026
✅ astro-stable-quickjs130026
✅ express-stable-node130026
✅ express-stable-quickjs130026
✅ fastify-stable-node130026
✅ fastify-stable-quickjs130026
✅ hono-stable-node130026
✅ hono-stable-quickjs130026
✅ nest-stable-node130026
✅ nest-stable-quickjs130026
✅ nextjs-turbopack-canary-node137019
✅ nextjs-turbopack-canary-quickjs137019
✅ nextjs-turbopack-stable-node15600
✅ nextjs-turbopack-stable-quickjs15600
✅ nextjs-webpack-canary-node137019
✅ nextjs-webpack-canary-quickjs137019
✅ nextjs-webpack-stable-node15600
✅ nextjs-webpack-stable-quickjs15600
✅ nitro-stable-node130026
✅ nitro-stable-quickjs130026
✅ nuxt-stable-node130026
✅ nuxt-stable-quickjs130026
✅ sveltekit-stable-node14907
✅ sveltekit-stable-quickjs14907
✅ tanstack-start-node130026
✅ tanstack-start-quickjs130026
✅ vite-stable-node130026
✅ vite-stable-quickjs130026

✅ 🪟 Windows

AppPassedFailedSkipped
✅ nextjs-turbopack-node15600
✅ nextjs-turbopack-quickjs15600

✅ vercel-multi-region

AppPassedFailedSkipped
✅ nextjs-turbopack2700

📋 View full workflow run

@github-actions

github-actionsBot commented Aug 13, 2026

Copy link
Copy Markdown
Contributor

📊 Workflow Benchmarks

commit b40e1aa · Thu, 13 Aug 2026 22:47:55 GMT · run logs

Backend: vercel · app: nextjs-turbopack

MetricScenarioBest (ms)P75 (ms)P90 (ms)P99 (ms)Samples
TTFSstep1261 (+585%) 🔻1377 🔴 (+24%) 🔻1485 🔴 (+22%) 🔻1712 🔴 (+7.6%)30
TTFSstream1289 (+444%) 🔻1339 🔴 (+21%) 🔻1382 🔴 (+22%) 🔻1409 🔴 (+17%) 🔻30
TTFShook + stream460 (+20%) 🔻1649 🔴 (+20%) 🔻1729 🔴 (+24%) 🔻1801 🔴 (+17%) 🔻30
Fan-out TTFSPromise.all(100 steps)8796 (-2.5%)9486 (-8.3%)10515 (-12%)14740 (-95%) 💚10
Fan-out TTLSPromise.all(100 steps)17503 (-1.4%)18992 (-7.4%)20565 (-70%) 💚24176 (-92%) 💚10
STSO1020 steps (inline)118 (-22%) 💚164 (-27%) 💚180 (-30%) 💚288 (-45%) 💚1019
WO1020 steps168616 (-26%) 💚168616 (-26%) 💚168616 (-26%) 💚168616 (-26%) 💚1
SLstream latency86 (-8.5%)126 🔴 (-16%) 💚146 🔴 (-9.3%)344 🔴 (-2.8%)30
SOstream overhead (text)105 (-33%) 💚161 (-46%) 💚167 (-62%) 💚218 (-85%) 💚30
SOstream overhead (structured)106 (-10%)164 (-60%) 💚174 (-69%) 💚296 (-88%) 💚30
📈 STSO distribution vs main (inline / queue-hop histograms)

1020 steps (inline)

Cumulative STSO time: main 228513ms → this run 168423ms (Δ -60090ms, -26%)

 100-150 ms ░░░░░░░░░░░░░░░░░░░┃ main 0 this 436 +436
150-200 ms ███████████████████████┃ main 512 this 536 +24
200-250 ms ┃████████████████ main 385 this 29 -356
250-300 ms ┃██ main 71 this 8 -63
300-350 ms ┃ main 22 this 5 -17
350-400 ms ┃ main 11 this 4 -7
400-450 ms ┃ main 5 this 0 -5
450-500 ms ┃ main 1 this 0 -1
500-550 ms ┃ main 4 this 0 -4
600-650 ms ┃ main 1 this 0 -1
650-700 ms ┃ main 1 this 0 -1
900-950 ms ┃ main 1 this 0 -1
1200-1250 ms ┃ main 1 this 0 -1
1600-1650 ms ┃ main 1 this 0 -1
2000-2050 ms ┃ main 1 this 0 -1
2400-2450 ms ┃ main 1 this 0 -1
6050-6100 ms ┃ main 1 this 0 -1
7900-7950 ms ┃ main 0 this 1 +1
ℹ️ Metric definitions & methodology

The collapsed STSO distribution section above buckets every step gap of the sequential-steps run (not a sampled window), split by whether the step ending the gap ran inline — in the same warm process as the step before it, so the gap is pure framework overhead — or after a queue-hop — the first step of a fresh process, which pays queue dispatch, client reinit and event-log replay. Bars overlay the two runs: is main, marks where this run lands, bridges the gap when this run has more samples in a bucket.

Best/P75/P90/P99 deltas compare against the most recent benchmark run on main at the time of this run. 🔻 flags a delta worse than +15%, 💚 one better than −15%.

Metrics — TTFS: time to first step body (in-deployment start() → first step body, deployment clocks) · Fan-out TTFS: fan-out time to first step (in-deployment start() → first of the parallel step bodies to complete) · Fan-out TTLS: fan-out time to last step (in-deployment start() → last of the parallel step bodies to complete, i.e. when the Promise.all resolves) · STSO: step-to-step overhead (gap between consecutive step bodies) · WO: workflow overhead (whole-run time outside step bodies, in-deployment anchored) · SL: stream latency (in-deployment write → read propagation, readAt - writtenAt) · SO: stream overhead (end-to-end write+consume time beyond the modelled generation window)

Scenarios — step: one trivial no-op step, no stream; no hooks, so the run stays in turbo mode (in-process fast path) · stream: one streaming step; no hooks, so the run stays in turbo mode (in-process fast path) · hook + stream: registers a hook before one step, which exits turbo mode (dispatch path) · 1020 steps: 1020 trivial sequential steps; STSO is measured between consecutive steps in the given step ranges, and WO is the whole-run overhead outside step bodies · Promise.all(100 steps): 100 trivial no-op steps started together in a single Promise.all; Fan-out TTFS is the first of them to complete and Fan-out TTLS the last, both from the in-deployment clientStart, so their gap is the spread the runtime adds across the fan-out · stream latency: parallel reader/writer steps on a dedicated stream; SL is the in-deployment write->read propagation (readAt - writtenAt) · stream overhead (text): writer streams 300 variable-length text token deltas paced at 100/s for 3s (a haiku-size LLM's token throughput) while a parallel reader drains the whole stream; SO is the end-to-end write+consume time beyond the 3s generation window (overhead/backpressure) · stream overhead (structured): same workload as stream overhead (text), but each delta is an AI-SDK-style structured object ({ type: 'text-delta', id, text }) instead of a raw string, so the SO gap vs the text scenario is the added serialization cost

🔴 marks a percentile over its target (within target is left unmarked). Targets (p75/p90/p99, ms) — TTFS 200/300/600 · SL 50/60/125 · SO 250/500/1000

All metrics are measured from deployment-side timestamps only. Runs are triggered by an in-deployment route that stamps the anchor (clientStart) right before start(), so the CI runner’s request and its path through api.vercel.com sit outside every measured window. TTFS = in-deployment start() → first step body (turbo uses the in-process fast path, non-turbo the dispatch path), and includes the VQS dispatch hop plus any /flow cold start. Fan-out TTFS/TTLS are the first and last step completions of a single Promise.all over trivial steps, from the same anchor, so the gap between the two rows is the spread the runtime adds across the fan-out. STSO/WO are measured between step bodies on the deployment. SL is measured inside the workflow (parallel reader/writer steps), so it no longer includes the api.vercel.com read path.

Cold starts are kept in the numbers on purpose — they are part of real bursty-workload latency. The workbench deployment cold-starts the /flow invocation for a large fraction of runs, inflating P75+; the Best column shows the fastest (warm-start) sample for comparison.

@github-actions

Copy link
Copy Markdown
Contributor

Sim World

Simulated world deterministic testing for races. Traces

🟠 Mint-ordered log — 6 fail of 41 total

log=mint-ordered · fence=per-spec

scenariooutcomeeventsvirtreplayviolations
smoke-no-stepscompleted30msok0
smoke-one-stepcompleted60msok0
hook-at-step-startedcompleted120msok0
hook-at-step-completedcompleted120msok0
hook-at-hook-createdcompleted120msok0
deadline-hook-winscompleted71.0hok0
deadline-expirescompleted71.0hok0
long-sleepcompleted1130.0dok0
hook-never-arrivesstalled30msskipped0
step-retries-twicecompleted102.0sok0
parallel-stepscompleted90msok0
hook-on-execution-statecompleted120msok0
peek-hook-before-branchcompleted120msok0
peek-hook-after-branchcompleted120msok0
peek-hook-at-registrationcompleted120msok0
race-hook-before-probecompleted120msok0
race-hook-after-probecompleted120msok0
race-duplicate-deliverycompleted130msok0
attr-hook-before-stepcompleted110msok0
attr-hook-after-stepcompleted110msok0
attr-from-step-bodycompleted130msok0
fork-hook-after-timeoutcompleted141.0mok0
fork-hook-before-timeoutcompleted141.0mok0
count-hook-after-timeoutcompleted171.0mok0
count-hook-before-timeoutcompleted201.0mok0
stale-read-step-count-forkcompleted171.0mMISMATCH1
stale-read-equal-step-countscompleted141.0mMISMATCH1
step-vs-step-forkcompleted120msMISMATCH1
step-vs-step-fork-fencedcompleted120msMISMATCH1
fence-catches-benign-directioncompleted125msok0
in-flight-before-decisioncompleted171.0mMISMATCH1
in-flight-before-decision-countedcompleted201.0mok0
in-flight-after-decisionfailed142.0mMISMATCH1
stale-read-step-count-fork-fencedcompleted201.0mok0
fork-hook-winscompleted131.0mok0
fork-timeout-winscompleted131.0mok0
unclaimed-payload-under-forkcompleted171.0mok0
claimed-payload-under-forkcompleted171.0mok0
writers-independent-step-bodiescompleted120msok0
writers-scripted-tempocompleted120msok0
cancel-mid-stepcancelled70msskipped0

Full trace: world-sim-mint.txt

🟢 Append-only log — 0 fail of 41 total

log=append-only · fence=per-spec

scenariooutcomeeventsvirtreplayviolations
smoke-no-stepscompleted30msok0
smoke-one-stepcompleted60msok0
hook-at-step-startedcompleted120msok0
hook-at-step-completedcompleted120msok0
hook-at-hook-createdcompleted120msok0
deadline-hook-winscompleted71.0hok0
deadline-expirescompleted71.0hok0
long-sleepcompleted1130.0dok0
hook-never-arrivesstalled30msskipped0
step-retries-twicecompleted102.0sok0
parallel-stepscompleted90msok0
hook-on-execution-statecompleted120msok0
peek-hook-before-branchcompleted120msok0
peek-hook-after-branchcompleted120msok0
peek-hook-at-registrationcompleted120msok0
race-hook-before-probecompleted120msok0
race-hook-after-probecompleted120msok0
race-duplicate-deliverycompleted130msok0
attr-hook-before-stepcompleted110msok0
attr-hook-after-stepcompleted110msok0
attr-from-step-bodycompleted130msok0
fork-hook-after-timeoutcompleted141.0mok0
fork-hook-before-timeoutcompleted141.0mok0
count-hook-after-timeoutcompleted171.0mok0
count-hook-before-timeoutcompleted201.0mok0
stale-read-step-count-forkcompleted201.0mok0
stale-read-equal-step-countscompleted141.0mok0
step-vs-step-forkcompleted120msok0
step-vs-step-fork-fencedcompleted120msok0
fence-catches-benign-directioncompleted125msok0
in-flight-before-decisioncompleted171.0mok0
in-flight-before-decision-countedcompleted171.0mok0
in-flight-after-decisioncompleted192.0mok0
stale-read-step-count-fork-fencedcompleted201.0mok0
fork-hook-winscompleted131.0mok0
fork-timeout-winscompleted131.0mok0
unclaimed-payload-under-forkcompleted171.0mok0
claimed-payload-under-forkcompleted171.0mok0
writers-independent-step-bodiescompleted120msok0
writers-scripted-tempocompleted120msok0
cancel-mid-stepcancelled70msskipped0

Full trace: world-sim-append-only.txt

@alangenfeld
alangenfeld marked this pull request as ready for review August 14, 2026 14:47
@alangenfeld
alangenfeld requested a review from a team as a code ownerAugust 14, 2026 14:47

@VaguelySeriousVaguelySerious left a comment

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The flaky annotation addition is great. We should still take care to check the flakes, and hopefully we'll get some more time soon to fix the actual tests, but this is great in the meantime

@alangenfeld
alangenfeld merged commit c041d3d into mainAug 14, 2026
293 of 297 checks passed
@alangenfeld
alangenfeld deleted the alangenfeld/e2e-ci-retry branch August 14, 2026 17:08
@github-actionsgithub-actionsBot mentioned this pull request Aug 14, 2026
@github-actions

Copy link
Copy Markdown
Contributor

Backport to stable failed for c041d3d due to a workflow error (backport job run).

This is usually an infrastructure problem (e.g. the configured AI model could not be found, an AI Gateway error, or an opencode crash) rather than a merge conflict. Check the job logs linked above for details.

Once the underlying issue is fixed, re-run the Backport to stable workflow manually via workflow_dispatch and paste this commit SHA into the ref input:

c041d3d231f8a75236311df56a68bd4ca104be22

pranaygp added a commit that referenced this pull request Aug 17, 2026
## Summary & Motivation
The canary e2e lanes (E2E Local Dev / Local Prod / Local Postgres for
`nextjs-turbopack` and `nextjs-webpack`, × node/quickjs) were pinned to
`16.3.0-canary.2`. This bumps the pin to the current canary,
`16.3.1-canary.17`, in all six spots in `tests.yml`.
Bumping the pin alone is not enough: fresh Next canaries are younger
than the repo's 48h `minimumReleaseAge` gate, so the "Setup canary"
`pnpm install --no-frozen-lockfile` fails with
`ERR_PNPM_NO_MATURE_MATCHING_VERSION` (reproduced locally with this
exact version). This adds `next` and its lockstep-published `@next/*`
companion packages (`@next/env`, `@next/swc-*`) to
`minimumReleaseAgeExclude` — both are Vercel-published, matching the
trust model of the existing exclusions (`@vercel/*`, `turbo`,
`esbuild`).
## Validation
The full e2e suite was run locally against `next@16.3.1-canary.17` on
all three worlds (staged tarball workbenches, same as CI's
`prepare-workbench-path`):
| World | App / mode | Result |
| --- | --- | --- |
| `world-local` | nextjs-webpack, dev server, node vm | `dev.test.ts` 5
passed / 4 skipped (incl. the HMR fuzz test) · `test:e2e` 137 passed /
19 skipped |
| `world-postgres` | nextjs-turbopack, prod build, node vm |
`local-build.test.ts` 13 passed · `test:e2e` 137 passed / 19 skipped¹ |
| `world-vercel` | nextjs-turbopack, preview deployment
(`dpl_3siALD2WyrsrdPFjmRY3ymwEHZiJ`), node vm | `test:e2e` 134 passed /
22 skipped |
¹ First pass failed the 3 `pages router` tests because the local harness
ran `local-build.test.ts` without `CI=true`, which deletes the
`workflow-sourcemap-warning-fixture` package the built output still
references (the test preserves it only when `CI=true`, exactly as its
comment warns). With the fixture preserved, all 3 pass. Not a canary
issue.
Note: the vercel world has no canary lane in CI (deployments build from
the committed stable pin), so the preview-deployment run above is the
only canary coverage it got.
## Context on current `main` CI
Recent `main` runs are red, but not because of the canary pin: the
dominant failure is the nextjs-webpack local-dev HMR fuzz race
(SDK-side, fix in flight in #3529), which hits the stable and canary
lanes alike; the rest are the known rotating vercel-prod flakes. The
latest completed `main` run (with #3530's e2e retry) passed every canary
lane and failed only the *stable* webpack dev lane.
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
Sign up for freeto join this conversation on GitHub. Already have an account? Sign in to comment

Labels

None yet

Projects

None yet

Development

Successfully merging this pull request may close these issues.

2 participants

@alangenfeld@VaguelySerious