') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); })(); [core] Replace never-picked-up e2e runs instead of failing whole tests by alangenfeld · Pull Request #3560 · vercel/workflow · GitHub
Skip to content

[core] Replace never-picked-up e2e runs instead of failing whole tests - #3560

Merged
alangenfeld merged 1 commit into
mainfrom
alangenfeld/e2e-start-watchdog
Aug 14, 2026
Merged

[core] Replace never-picked-up e2e runs instead of failing whole tests#3560
alangenfeld merged 1 commit into
mainfrom
alangenfeld/e2e-start-watchdog

Conversation

@alangenfeld

Copy link
Copy Markdown
Collaborator

Summary & Motivation

The e2e start() wrappers now poll the new run until it leaves pending. A run still pending after WORKFLOW_E2E_PICKUP_BUDGET_MS (default 15s) has executed no workflow code, so it is abandoned and replaced in place and the test continues — one replacement, with the CI-level retry still the backstop if that one stalls too.

Each replacement is recorded to an e2e-infra-*.json sidecar that every e2e job uploads, and the aggregation script renders it as an "Infra Events" section in the step summary and PR comment, so clustered timestamps read as a backend blip rather than as unrelated flaky tests.

Test Plan

Unit tests cover the pickup watchdog; a local nextjs-turbopack run exercised both the clean path and, with a forced 1ms budget, the replacement path end to end, and the aggregation script was smoke-tested against synthetic sidecars in both modes.

@changeset-bot

changeset-botBot commented Aug 14, 2026

Copy link
Copy Markdown

🦋 Changeset detected

Latest commit: 5c7f822

The changes in this PR will be included in the next version bump.

This PR includes changesets to release 16 packages
NameType
@workflow/corePatch
@workflow/buildersPatch
@workflow/cliPatch
@workflow/nextPatch
@workflow/nitroPatch
@workflow/vitestPatch
@workflow/web-sharedPatch
@workflow/webPatch
workflowPatch
@workflow/world-testingPatch
@workflow/astroPatch
@workflow/nestPatch
@workflow/rollupPatch
@workflow/sveltekitPatch
@workflow/vitePatch
@workflow/nuxtPatch

Not sure what this means? Click here to learn what changesets are.

Click here if you're a maintainer who wants to add another changeset to this PR

@vercel

vercelBot commented Aug 14, 2026

Copy link
Copy Markdown
Contributor

The latest updates on your projects. Learn more about Vercel for GitHub.

ProjectDeploymentActionsUpdated (UTC)
example-nextjs-workflow-turbopackReadyReadyPreviewAug 14, 2026 9:19pm
example-nextjs-workflow-webpackReadyReadyPreviewAug 14, 2026 9:19pm
example-workflowReadyReadyPreviewAug 14, 2026 9:19pm
workbench-astro-workflowReadyReadyPreviewAug 14, 2026 9:19pm
workbench-express-workflowReadyReadyPreviewAug 14, 2026 9:19pm
workbench-fastify-workflowReadyReadyPreviewAug 14, 2026 9:19pm
workbench-hono-workflowReadyReadyPreviewAug 14, 2026 9:19pm
workbench-nestjs-workflowReadyReadyPreviewAug 14, 2026 9:19pm
workbench-nitro-workflowReadyReadyPreviewAug 14, 2026 9:19pm
workbench-nuxt-workflowReadyReadyPreviewAug 14, 2026 9:19pm
workbench-python-workflowReadyReadyPreviewAug 14, 2026 9:19pm
workbench-sveltekit-workflowReadyReadyPreviewAug 14, 2026 9:19pm
workbench-tanstack-start-workflowReadyReadyPreviewAug 14, 2026 9:19pm
workbench-vite-workflowReadyReadyPreviewAug 14, 2026 9:19pm
workflow-docsReadyReadyPreview, v0Aug 14, 2026 9:19pm
workflow-swc-playgroundReadyReadyPreviewAug 14, 2026 9:19pm
workflow-tarballsReadyReadyPreviewAug 14, 2026 9:19pm
workflow-webReadyReadyPreviewAug 14, 2026 9:19pm

@github-actions

github-actionsBot commented Aug 14, 2026

Copy link
Copy Markdown
Contributor

🧪 E2E Test Results

All tests passed

⚠️ Flaky E2E Tests (passed on retry)

These tests failed at least once and passed on a retry. A recurring entry here is a real race worth investigating.

  • addTenWorkflow via pages router (nextjs-webpack)

🛠 Infra Events (absorbed by the harness)

Platform anomalies the e2e harness detected and worked around (e.g. a run the queue never picked up, replaced by a fresh run). Clustered timestamps indicate a backend blip; a steady drip indicates a platform issue worth escalating.

  • run-pickup-stall · addTenWorkflow (tanstack-start) · at 21:22:35Z · abandoned wrun_01M012F2JR85PVCBAWSNGN1KE4

E2E Test Summary

Summary
PassedFailedSkippedTotal
✅ ▲ Vercel Production347407384212
✅ 💻 Local Development381005584368
✅ 📦 Local Production381005584368
✅ 🐘 Local Postgres381005584368
✅ 🪟 Windows31200312
✅ 🌐 Cross-language Conformance90128137
✅ vercel-multi-region270027
Total152520254017792
Details by Category

✅ ▲ Vercel Production

AppPassedFailedSkipped
✅ astro-node128028
✅ astro-quickjs128028
✅ example-node128028
✅ example-quickjs128028
✅ express-node128028
✅ express-quickjs128028
✅ fastify-node128028
✅ fastify-quickjs128028
✅ hono-node128028
✅ hono-quickjs128028
✅ nest-node128028
✅ nest-quickjs128028
✅ nextjs-turbopack-node15303
✅ nextjs-turbopack-quickjs15303
✅ nextjs-webpack-node15303
✅ nextjs-webpack-quickjs15303
✅ nitro-node128028
✅ nitro-quickjs128028
✅ nuxt-node128028
✅ nuxt-quickjs128028
✅ python-node80148
✅ sveltekit-node14709
✅ sveltekit-quickjs14709
✅ tanstack-start-node128028
✅ tanstack-start-quickjs128028
✅ vite-node128028
✅ vite-quickjs128028

✅ 💻 Local Development

AppPassedFailedSkipped
✅ astro-stable-node130026
✅ astro-stable-quickjs130026
✅ express-stable-node130026
✅ express-stable-quickjs130026
✅ fastify-stable-node130026
✅ fastify-stable-quickjs130026
✅ hono-stable-node130026
✅ hono-stable-quickjs130026
✅ nest-stable-node130026
✅ nest-stable-quickjs130026
✅ nextjs-turbopack-canary-node137019
✅ nextjs-turbopack-canary-quickjs137019
✅ nextjs-turbopack-stable-node15600
✅ nextjs-turbopack-stable-quickjs15600
✅ nextjs-webpack-canary-node137019
✅ nextjs-webpack-canary-quickjs137019
✅ nextjs-webpack-stable-node15600
✅ nextjs-webpack-stable-quickjs15600
✅ nitro-stable-node130026
✅ nitro-stable-quickjs130026
✅ nuxt-stable-node130026
✅ nuxt-stable-quickjs130026
✅ sveltekit-stable-node14907
✅ sveltekit-stable-quickjs14907
✅ tanstack-start-node130026
✅ tanstack-start-quickjs130026
✅ vite-stable-node130026
✅ vite-stable-quickjs130026

✅ 📦 Local Production

AppPassedFailedSkipped
✅ astro-stable-node130026
✅ astro-stable-quickjs130026
✅ express-stable-node130026
✅ express-stable-quickjs130026
✅ fastify-stable-node130026
✅ fastify-stable-quickjs130026
✅ hono-stable-node130026
✅ hono-stable-quickjs130026
✅ nest-stable-node130026
✅ nest-stable-quickjs130026
✅ nextjs-turbopack-canary-node137019
✅ nextjs-turbopack-canary-quickjs137019
✅ nextjs-turbopack-stable-node15600
✅ nextjs-turbopack-stable-quickjs15600
✅ nextjs-webpack-canary-node137019
✅ nextjs-webpack-canary-quickjs137019
✅ nextjs-webpack-stable-node15600
✅ nextjs-webpack-stable-quickjs15600
✅ nitro-stable-node130026
✅ nitro-stable-quickjs130026
✅ nuxt-stable-node130026
✅ nuxt-stable-quickjs130026
✅ sveltekit-stable-node14907
✅ sveltekit-stable-quickjs14907
✅ tanstack-start-node130026
✅ tanstack-start-quickjs130026
✅ vite-stable-node130026
✅ vite-stable-quickjs130026

✅ 🐘 Local Postgres

AppPassedFailedSkipped
✅ astro-stable-node130026
✅ astro-stable-quickjs130026
✅ express-stable-node130026
✅ express-stable-quickjs130026
✅ fastify-stable-node130026
✅ fastify-stable-quickjs130026
✅ hono-stable-node130026
✅ hono-stable-quickjs130026
✅ nest-stable-node130026
✅ nest-stable-quickjs130026
✅ nextjs-turbopack-canary-node137019
✅ nextjs-turbopack-canary-quickjs137019
✅ nextjs-turbopack-stable-node15600
✅ nextjs-turbopack-stable-quickjs15600
✅ nextjs-webpack-canary-node137019
✅ nextjs-webpack-canary-quickjs137019
✅ nextjs-webpack-stable-node15600
✅ nextjs-webpack-stable-quickjs15600
✅ nitro-stable-node130026
✅ nitro-stable-quickjs130026
✅ nuxt-stable-node130026
✅ nuxt-stable-quickjs130026
✅ sveltekit-stable-node14907
✅ sveltekit-stable-quickjs14907
✅ tanstack-start-node130026
✅ tanstack-start-quickjs130026
✅ vite-stable-node130026
✅ vite-stable-quickjs130026

✅ 🪟 Windows

AppPassedFailedSkipped
✅ nextjs-turbopack-node15600
✅ nextjs-turbopack-quickjs15600

✅ 🌐 Cross-language Conformance

AppPassedFailedSkipped
✅ python90128

✅ vercel-multi-region

AppPassedFailedSkipped
✅ nextjs-turbopack2700

📋 View full workflow run

@github-actions

github-actionsBot commented Aug 14, 2026

Copy link
Copy Markdown
Contributor

📊 Workflow Benchmarks

commit 5c7f822 · Fri, 14 Aug 2026 21:40:40 GMT · run logs

Backend: vercel · app: nextjs-turbopack

MetricScenarioBest (ms)P75 (ms)P90 (ms)P99 (ms)Samples
TTFSstep1256 (+23%) 🔻1496 🔴 (+34%) 🔻1542 🔴 (+14%)1863 🔴 (+2.5%)30
TTFSstream458 (-54%) 💚1480 🔴 (+37%) 🔻1507 🔴 (+38%) 🔻1539 🔴 (+20%) 🔻30
TTFShook + stream396 (-68%) 💚1964 🔴 (+44%) 🔻2108 🔴 (+46%) 🔻2792 🔴 (+76%) 🔻30
Fan-out TTFSPromise.all(100 steps)9345 (+3.4%)10862 (+11%)10924 (+7.3%)15653 (+13%)10
Fan-out TTLSPromise.all(100 steps)18509 (+2.3%)20037 (-15%) 💚20543 (-17%) 💚25802 (+3.2%)10
STSO1020 steps (inline)131 (+10%)168 (-60%) 💚184 (-62%) 💚298 (-56%) 💚1019
WO1020 steps169100 (-53%) 💚169100 (-53%) 💚169100 (-53%) 💚169100 (-53%) 💚1
CRTTfirst chunk (pooled)123 (+45%) 🔻167 (+45%) 🔻218 (+31%) 🔻245 (+15%) 🔻28

Streams

Scenariowr c/srd c/swr KiB/srd KiB/sCRTT 1stp75p90p99CDV maxiters
paced control (100/s, 60B)100 (±0%)102 (+1%)5 (±0%)5.1 (+1%)151 (+33%)209 (-34%)443 (-49%)639 (-83%)184 (+41%)10
size sweep (100/s, 160B-12KB)100 (±0%)100 (±0%)334 (±0%)335 (±0%)152 (+49%)175 (+12%)228 (±0%)365 (-59%)119 (-22%)10
replay gateway-gpt-5.4-nano-2000t (1x)89.2 (±0%)89.2 (±0%)16.2 (±0%)16.2 (±0%)148 (+11%)1048 (+700%)3562 (+1526%)5092 (+537%)541 (-15%)3
replay eve-gpt-5.6-sol-2000t (1x)54.7 (±0%)54.7 (±0%)355 (±0%)355 (±0%)146 (+30%)173 (+38%)231 (+46%)860 (+228%)495 (+77%)2
replay eve-gpt-5.6-sol-2000t (2x)109 (±0%)109 (±0%)710 (±0%)709 (±0%)178 (+80%)349 (-62%)495 (-81%)1420 (-68%)754 (+35%)3
📈 STSO distribution vs main (inline / queue-hop histograms)

1020 steps (inline)

Cumulative STSO time: main 359456ms → this run 168861ms (Δ -190595ms, -53%)

100-150 ms █░┃ main 18 this 98 +80
150-200 ms █░░░░░░░░░░░░░░░░░░░░░░┃ main 48 this 862 +814
200-250 ms ┃█ main 80 this 37 -43
250-300 ms ┃████ main 164 this 12 -152
300-350 ms ┃█████ main 221 this 7 -214
350-400 ms ┃████ main 188 this 2 -186
400-450 ms ┃███ main 129 this 1 -128
450-500 ms ┃█ main 86 this 0 -86
500-550 ms ┃ main 44 this 0 -44
550-600 ms ┃ main 15 this 0 -15
600-650 ms ┃ main 13 this 0 -13
650-700 ms ┃ main 6 this 0 -6
700-750 ms ┃ main 2 this 0 -2
750-800 ms ┃ main 4 this 0 -4
📈 CRTT drill-down vs main (RTT distributions & profiles)
variant RTT 1ms→5s+ avg p50 p90 p99 n
control ······▂█▂▁··· 167.1 (-55%) 136 (+36%) 443 (-49%) 639 (-83%) 3000
sweep ······▁█▂···· 147.4 (+15%) 141 (+28%) 228 (±0%) 365 (-59%) 3000
gw 1x ·····▁▂█▂▁▁▁▁ 407 (+215%) 144 (+43%) 3562 (+1526%) 5092 (+537%) 5295
eve 1x ·····▁▂█▂▁▁·· 150.6 (+41%) 129 (+32%) 231 (+46%) 860 (+228%) 5186
eve 2x ······▁█▆▁▁·· 242.2 (-34%) 176 (+40%) 495 (-81%) 1420 (-68%) 7779

RTT over stream progress (avg per tenth of stream, bars scaled min→max):

control ▇▅█▇▃▁▂▂▁▁ 126–234ms
sweep █▁▁▂▂▄▆▂▃▂ 132–184ms
gw 1x ▁▁▁▁█▆▃▁▁▁ 131–1424ms
eve 1x █▁▃▂▂▃▃▄▂▃ 118–224ms
eve 2x ▄▆█▁▁▂▃▄▃▂ 166–383ms

RTT by chunk size (avg per log size bin, ~160B → ~12KB serialized, bars scaled min→max):

sweep ▁▄█▂▄▆▂ 147–148ms

Delivery jitter over stream progress (avg positive CDV per tenth of stream, bars scaled min→max):

control ▁▆█▃▃▃▂▂▂▂ 38–67ms
sweep ▁▅▆▆█▇█▇▇▆ 36–68ms
gw 1x ▂▂▁▁█▁▁▂▁▂ 38–125ms
eve 1x █▃▂▂▂▆▃▄▁▅ 25–34ms
eve 2x ▆█▅▂▅▃▂▁▃▅ 23–43ms
📜 Previous results (2)

c06fbfc

Fri, 14 Aug 2026 20:42:43 GMT · run logs

vercel / nextjs-turbopack

MetricScenarioBest (ms)P75 (ms)P90 (ms)P99 (ms)Samples
TTFSstep254 (+15%) 🔻1390 🔴 (+25%) 🔻1418 🔴 (+20%) 🔻1456 🔴 (+21%) 🔻30
TTFSstream253 (+6.3%)1398 🔴 (+25%) 🔻1409 🔴 (+23%) 🔻1730 🔴 (+39%) 🔻30
TTFShook + stream367 (-26%) 💚1668 🔴 (+26%) 🔻1717 🔴 (+24%) 🔻2024 🔴 (+45%) 🔻30
Fan-out TTFSPromise.all(100 steps)9124 (±0%)10850 (+7.5%)10890 (+6.4%)11101 (+7.1%)10
Fan-out TTLSPromise.all(100 steps)18192 (+3.0%)20215 (+8.0%)21567 (+15%)24878 (+25%) 🔻10
STSO1020 steps (inline)145 (+2.8%)461 (+118%) 🔻515 (+116%) 🔻670 (+86%) 🔻1018
STSO1020 steps (queue-hop)32143214321432141
WO1020 steps414013 (+103%) 🔻414013 (+103%) 🔻414013 (+103%) 🔻414013 (+103%) 🔻1
SLstream latency109 (+14%)194 🔴 (+20%) 🔻346 🔴 (+83%) 🔻857 🔴 (+131%) 🔻30
SOstream overhead (text)127 (-15%) 💚357 🔴 (+43%) 🔻510 🔴 (-20%) 💚951 (-46%) 💚30
SOstream overhead (structured)134 (-13%)281 🔴 (-8.5%)698 🔴 (+87%) 🔻844 (+24%) 🔻30

f5d8217

Fri, 14 Aug 2026 19:20:26 GMT · run logs

vercel / nextjs-turbopack

MetricScenarioBest (ms)P75 (ms)P90 (ms)P99 (ms)Samples
TTFSstep228 (-78%) 💚1417 🔴 (+20%) 🔻1436 🔴 (+13%)1510 🔴 (+15%) 🔻30
TTFSstream248 (-2.4%)1431 🔴 (+24%) 🔻1485 🔴 (+23%) 🔻1733 🔴 (+38%) 🔻30
TTFShook + stream465 (-13%)1658 🔴 (+12%)1679 🔴 (+8.5%)2047 🔴 (+18%) 🔻30
Fan-out TTFSPromise.all(100 steps)9107 (+3.1%)10600 (+12%)10715 (+4.6%)10822 (+3.9%)10
Fan-out TTLSPromise.all(100 steps)17891 (+3.1%)19587 (+7.2%)19833 (+3.2%)20781 (+3.0%)10
STSO1020 steps (inline)137 (+2.2%)202 (-14%)234 (-18%) 💚342 (-51%) 💚1019
WO1020 steps196813 (-14%)196813 (-14%)196813 (-14%)196813 (-14%)1
SLstream latency100 (-4.8%)145 🔴 (-28%) 💚157 🔴 (-65%) 💚857 🔴 (+5.9%)30
SOstream overhead (text)166 (+23%) 🔻462 🔴 (+56%) 🔻629 🔴 (+15%)965 (-25%) 💚30
SOstream overhead (structured)140 (-11%)360 🔴 (+33%) 🔻655 🔴 (+78%) 🔻3838 🔴 (+412%) 🔻30
ℹ️ Metric definitions & methodology

Streams: writer/reader sustained rates (steady window, 10% trimmed each side), first-chunk RTT (the stream-open path, before any buffering/backpressure), CRTT percentiles, and worst delivery stall (CDV max). Cells are medians across iterations; per-run values in the artifacts. No 🔴/🟢 marks until targets attach.

The collapsed STSO distribution section above buckets every step gap, split inline (same warm process — pure framework overhead) vs queue-hop (fresh process — dispatch, reinit, replay). = main, = this run, = fill.

The collapsed CRTT drill-down: per-variant RTT histograms (fixed log bins, · = empty) and mean RTT/positive-CDV profile lines over stream progress and chunk size. Histograms, avgs, and profiles merge exactly across runs; p50–p99 are percentile-of-percentiles. Per-index rows live in the artifacts.

Best/P75/P90/P99 deltas compare against the most recent benchmark run on main at the time of this run. 🔻 flags a delta worse than +15%, 💚 one better than −15%.

Metrics — TTFS: time to first step body (in-deployment start() → first step body) · Fan-out TTFS: fan-out time to first step (in-deployment start() → first of the parallel step bodies to complete) · Fan-out TTLS: fan-out time to last step (in-deployment start() → last of the parallel step bodies to complete, i.e. when the Promise.all resolves) · STSO: step-to-step overhead (gap between consecutive step bodies) · WO: workflow overhead (whole-run time outside step bodies, in-deployment anchored) · CRTT: chunk round-trip time (per-chunk write → read latency, one clock domain: deployment → stream backend → same deployment) · CDV: chunk delay variation / delivery jitter (inter-arrival gap minus inter-write gap per seq-adjacent pair; skew-free; the row is each run's MAX positive value, so one stall moves it)

Scenarios — step: one trivial no-op step, no stream; no hooks, so the run stays in turbo mode (in-process fast path) · stream: one streaming step; no hooks, so the run stays in turbo mode (in-process fast path) · hook + stream: registers a hook before one step, which exits turbo mode (dispatch path) · 1020 steps: 1020 trivial sequential steps; STSO is measured between consecutive steps in the given step ranges, and WO is the whole-run overhead outside step bodies · Promise.all(100 steps): 100 trivial no-op steps started together in a single Promise.all; Fan-out TTFS is the first of them to complete and Fan-out TTLS the last, both from the in-deployment clientStart, so their gap is the spread the runtime adds across the fan-out · paced control (100/s, 60B): the control: 300 tiny (~60B) deltas metronome-paced at 100/s — zero workload structure, so it reads the transport floor and flush cadence, and disambiguates transport-wide vs workload-specific when a replay row moves · size sweep (100/s, 160B-12KB): same pacing as the control with deltas padded in rotation across seven log-spaced sizes (~160B–12KB) — rotation decouples size from stream position, so it isolates whether chunk size causes latency · replay gateway-gpt-5.4-nano-2000t (1x): raw provider SSE cadence captured at the AI gateway boundary (gpt-5.4-nano, the most popular gateway model; per-token deltas p50 208B = the modal production chunk size), replayed exactly as measured — the typical customer's workload; its CDV is the typical customer's real delivery jitter · replay eve-gpt-5.6-sol-2000t (1x): a captured eve turn (gpt-5.6-sol, the most-used demanding eve model; ~2000 output tokens = production p50 turn length) replayed exactly as measured — eve's envelope protocol re-ships the cumulative message so sizes ramp 142B→13KB; the demanding outlier tenant's reality · replay eve-gpt-5.6-sol-2000t (2x): the same eve capture at 2x — the headroom/stress row; real fast-tier models emit the same chunk sizes at proportionally higher rate, so time compression is a faithful speed model · first chunk (pooled): every run's seq-0 RTT pooled across all stream scenarios — the first chunk precedes any workload differentiation, so pooling samples one shared stream-open path with exact percentiles

Replay cadences (semantic sha256) — eve-gpt-5.6-sol-2000teaf22f5946e7c61f3c65c7006d550df180cfabd4e706254a09f22aec0cfb420d · gateway-gpt-5.4-nano-2000t6f24ac518b6b83ff1d0e85a5fe78230db192716d66a7fc6b2fe022752001d041

🔴 marks a percentile over its target (within target is left unmarked). Targets (p75/p90/p99, ms) — TTFS 200/300/600

All timestamps are deployment-side; runs are triggered in-deployment, so the CI runner and api.vercel.com sit outside every measured window. TTFS = start() → first step body (includes dispatch + any cold start); Fan-out TTFS/TTLS = first/last step completion of one Promise.all from the same anchor (the gap is the runtime’s fan-out spread); STSO/WO between step bodies; CRTT inside the workflow (excludes the api.vercel.com read path).

Cold starts stay in the numbers (real bursty-workload latency, inflates P75+); Best is the warm floor.

@github-actions

github-actionsBot commented Aug 14, 2026

Copy link
Copy Markdown
Contributor

Sim World

Simulated world deterministic testing for races. Traces

🟠 Mint-ordered log — 3 fail of 41 total

log=mint-ordered · fence=per-spec

scenariooutcomeeventsvirtreplayviolations
smoke-no-stepscompleted30msok0
smoke-one-stepcompleted60msok0
hook-at-step-startedcompleted120msok0
hook-at-step-completedcompleted120msok0
hook-at-hook-createdcompleted120msok0
deadline-hook-winscompleted71.0hok0
deadline-expirescompleted71.0hok0
long-sleepcompleted1130.0dok0
hook-never-arrivesstalled30msskipped0
step-retries-twicecompleted102.0sok0
parallel-stepscompleted90msok0
hook-on-execution-statecompleted120msok0
peek-hook-before-branchcompleted120msok0
peek-hook-after-branchcompleted120msok0
peek-hook-at-registrationcompleted120msok0
race-hook-before-probecompleted120msok0
race-hook-after-probecompleted120msok0
race-duplicate-deliverycompleted130msok0
attr-hook-before-stepcompleted110msok0
attr-hook-after-stepcompleted110msok0
attr-from-step-bodycompleted130msok0
fork-hook-after-timeoutcompleted141.0mok0
fork-hook-before-timeoutcompleted141.0mok0
count-hook-after-timeoutcompleted171.0mok0
count-hook-before-timeoutcompleted201.0mok0
stale-read-step-count-forkcompleted201.0mok0
stale-read-equal-step-countscompleted141.0mok0
step-vs-step-forkcompleted120msok0
step-vs-step-fork-fencedcompleted120msok0
fence-catches-benign-directioncompleted125msok0
in-flight-before-decisionfailed91.0mMISMATCH1
in-flight-before-decision-countedfailed91.0mMISMATCH1
in-flight-after-decisionfailed91.0mMISMATCH1
stale-read-step-count-fork-fencedcompleted201.0mok0
fork-hook-winscompleted131.0mok0
fork-timeout-winscompleted131.0mok0
unclaimed-payload-under-forkcompleted171.0mok0
claimed-payload-under-forkcompleted171.0mok0
writers-independent-step-bodiescompleted120msok0
writers-scripted-tempocompleted120msok0
cancel-mid-stepcancelled70msskipped0

Full trace: world-sim-mint.txt

🟢 Append-only log — 0 fail of 41 total

log=append-only · fence=per-spec

scenariooutcomeeventsvirtreplayviolations
smoke-no-stepscompleted30msok0
smoke-one-stepcompleted60msok0
hook-at-step-startedcompleted120msok0
hook-at-step-completedcompleted120msok0
hook-at-hook-createdcompleted120msok0
deadline-hook-winscompleted71.0hok0
deadline-expirescompleted71.0hok0
long-sleepcompleted1130.0dok0
hook-never-arrivesstalled30msskipped0
step-retries-twicecompleted102.0sok0
parallel-stepscompleted90msok0
hook-on-execution-statecompleted120msok0
peek-hook-before-branchcompleted120msok0
peek-hook-after-branchcompleted120msok0
peek-hook-at-registrationcompleted120msok0
race-hook-before-probecompleted120msok0
race-hook-after-probecompleted120msok0
race-duplicate-deliverycompleted130msok0
attr-hook-before-stepcompleted110msok0
attr-hook-after-stepcompleted110msok0
attr-from-step-bodycompleted130msok0
fork-hook-after-timeoutcompleted141.0mok0
fork-hook-before-timeoutcompleted141.0mok0
count-hook-after-timeoutcompleted171.0mok0
count-hook-before-timeoutcompleted201.0mok0
stale-read-step-count-forkcompleted201.0mok0
stale-read-equal-step-countscompleted141.0mok0
step-vs-step-forkcompleted120msok0
step-vs-step-fork-fencedcompleted120msok0
fence-catches-benign-directioncompleted125msok0
in-flight-before-decisioncompleted171.0mok0
in-flight-before-decision-countedcompleted171.0mok0
in-flight-after-decisioncompleted192.0mok0
stale-read-step-count-fork-fencedcompleted201.0mok0
fork-hook-winscompleted131.0mok0
fork-timeout-winscompleted131.0mok0
unclaimed-payload-under-forkcompleted171.0mok0
claimed-payload-under-forkcompleted171.0mok0
writers-independent-step-bodiescompleted120msok0
writers-scripted-tempocompleted120msok0
cancel-mid-stepcancelled70msskipped0

Full trace: world-sim-append-only.txt

@alangenfeld
alangenfeld merged commit 234d3dd into mainAug 14, 2026
172 checks passed
@alangenfeld
alangenfeld deleted the alangenfeld/e2e-start-watchdog branch August 14, 2026 21:48
@github-actionsgithub-actionsBot mentioned this pull request Aug 14, 2026
@github-actions

Copy link
Copy Markdown
Contributor

No backport to stable for 234d3dd (AI decision).

This is additive e2e-harness tooling rather than a fix for a broken or flaky test on stable: it introduces a new pickup watchdog with a new config flag (WORKFLOW_E2E_PICKUP_BUDGET_MS), new exported harness helpers, a new e2e-infra-*.json sidecar, and a new "Infra Events" reporting section — capability, not repair, with the pre-existing CI-level retry still the backstop. It also builds on main-only harness infrastructure: stable's .github/scripts/aggregate-e2e-results.js has no flaky-sidecar loading/rendering (loadFlaky/renderFlakySection) for the new infra section to sit alongside, and packages/core/e2e/utils.test.ts does not exist there at all, so it would need adaptation rather than a straight application.

To override, re-run the Backport to stable workflow manually via workflow_dispatch and paste this commit SHA into the ref input:

234d3dd7b852129e189d321314c4f749f12711d8

Sign up for freeto join this conversation on GitHub. Already have an account? Sign in to comment

Labels

None yet

Projects

None yet

Development

Successfully merging this pull request may close these issues.

2 participants

@alangenfeld@VaguelySerious