diff --git a/.changeset/e2e-ci-retry.md b/.changeset/e2e-ci-retry.md new file mode 100644 index 0000000000..ba0a9e2aa8 --- /dev/null +++ b/.changeset/e2e-ci-retry.md @@ -0,0 +1,5 @@ +--- +'@workflow/core': patch +--- + +Retry e2e tests once in CI and surface retried-then-passed tests as flaky in annotations and the PR results comment. diff --git a/.github/scripts/aggregate-e2e-results.js b/.github/scripts/aggregate-e2e-results.js index 64bd9522dd..840ac0c3a8 100644 --- a/.github/scripts/aggregate-e2e-results.js +++ b/.github/scripts/aggregate-e2e-results.js @@ -53,6 +53,7 @@ function findResultFiles(dir) { return findJsonFiles(dir, 'e2e-', [ 'e2e-metadata-', 'e2e-failures-', + 'e2e-flaky-', 'e2e-diagnostics-', 'e2e-runtime-logs-', ]); @@ -147,6 +148,79 @@ function loadFailures(dir) { return failures; } +// Load flaky-test sidecar files (tests that passed only after a retry, +// written by github-reporter). Grouped per app; the same test flaking in +// several jobs for one app is collapsed into a single entry with an +// occurrence count so the section stays scannable. +function loadFlaky(dir) { + // Map of `${app}\u0000${testName}` -> { app, testName, retryCount, occurrences } + const flaky = new Map(); + const files = findJsonFiles(dir, 'e2e-flaky-'); + + for (const file of files) { + const basename = path.basename(file, '.json'); + const match = basename.match(/^e2e-flaky-(.+)-(?:vercel|local)$/); + const app = match ? match[1] : 'unknown'; + try { + const entries = JSON.parse(fs.readFileSync(file, 'utf-8')); + for (const entry of entries) { + if (!entry.testName) continue; + const key = `${app}\u0000${entry.testName}`; + const existing = flaky.get(key); + if (existing) { + existing.occurrences++; + existing.retryCount = Math.max( + existing.retryCount, + entry.retryCount || 1 + ); + } else { + flaky.set(key, { + app, + testName: entry.testName, + retryCount: entry.retryCount || 1, + occurrences: 1, + }); + } + } + } catch (_e) { + // Skip invalid files + } + } + + return [...flaky.values()]; +} + +// Render the flaky-tests section shared by the PR comment and the per-job +// step summary. Retried-to-green tests would otherwise be invisible — the +// job is green — so this is the only place a recurring race stays visible. +function renderFlakySection(flakyTests) { + if (flakyTests.length === 0) return; + + console.log('### ⚠️ Flaky E2E Tests (passed on retry)\n'); + console.log( + '_These tests failed at least once and passed on a retry. A recurring entry here is a real race worth investigating._\n' + ); + + const sorted = [...flakyTests].sort( + (a, b) => + b.occurrences - a.occurrences || a.testName.localeCompare(b.testName) + ); + const collapse = sorted.length >= 10; + if (collapse) { + console.log('
'); + console.log(`${sorted.length} flaky tests\n`); + } + for (const test of sorted) { + const jobs = + test.occurrences > 1 ? ` — flaked in ${test.occurrences} jobs` : ''; + console.log(`- \`${test.testName}\` (${test.app})${jobs}`); + } + console.log(''); + if (collapse) { + console.log('
\n'); + } +} + // vitest's JSON reporter serializes only error stacks. For test timeouts the // stack is the task-collection stack ("Error: STACK_TRACE_ERROR ..."), which // carries no information about the failure. The github-reporter failures @@ -358,7 +432,7 @@ function aggregateByCategory(files) { } // Render markdown summary for single job (step summary) -function renderSingleJobSummary(summary) { +function renderSingleJobSummary(summary, flakyTests = []) { const total = summary.totalPassed + summary.totalFailed + summary.totalSkipped; const statusEmoji = summary.totalFailed > 0 ? '❌' : '✅'; @@ -397,6 +471,8 @@ function renderSingleJobSummary(summary) { } } + renderFlakySection(flakyTests); + // Results by file if (summary.fileResults.length > 1) { console.log('
'); @@ -446,7 +522,8 @@ function renderAggregatedSummary( overallSummary, metadata, diagnostics, - failures + failures, + flakyTests ) { const total = overallSummary.totalPassed + @@ -556,6 +633,8 @@ function renderAggregatedSummary( } } + renderFlakySection(flakyTests); + // Everything else lives under one collapsible summary section. console.log('### E2E Test Summary\n'); @@ -624,13 +703,15 @@ if (mode === 'aggregate') { const metadata = loadMetadata(resultsDir); const diagnostics = loadDiagnostics(resultsDir); const failures = loadFailures(resultsDir); + const flakyTests = loadFlaky(resultsDir); enrichFailedTestMessages(overallSummary.allFailedTests, failures); renderAggregatedSummary( categories, overallSummary, metadata, diagnostics, - failures + failures, + flakyTests ); // Exit with non-zero if any tests failed @@ -640,7 +721,7 @@ if (mode === 'aggregate') { } else { const summary = aggregateResults(resultFiles); enrichFailedTestMessages(summary.allFailedTests, loadFailures(resultsDir)); - renderSingleJobSummary(summary); + renderSingleJobSummary(summary, loadFlaky(resultsDir)); // Exit with non-zero if any tests failed if (summary.totalFailed > 0) { diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 8e63495fa1..7d45584008 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -495,6 +495,7 @@ jobs: e2e-vercel-prod-${{ matrix.app.name }}-${{ matrix.vm }}.json e2e-metadata-${{ matrix.app.name }}-vercel.json e2e-failures-${{ matrix.app.name }}-vercel.json + e2e-flaky-${{ matrix.app.name }}-vercel.json e2e-diagnostics-${{ matrix.app.name }}-vercel.json e2e-runtime-logs-${{ matrix.app.name }}-vercel.json retention-days: 7 @@ -599,6 +600,7 @@ jobs: e2e-vercel-multi-region-nextjs-turbopack.json e2e-metadata-nextjs-turbopack-vercel.json e2e-failures-nextjs-turbopack-vercel.json + e2e-flaky-nextjs-turbopack-vercel.json e2e-diagnostics-nextjs-turbopack-vercel.json e2e-runtime-logs-nextjs-turbopack-vercel.json retention-days: 7 @@ -778,6 +780,7 @@ jobs: e2e-vercel-ws-transport-${{ matrix.app.name }}.json e2e-metadata-${{ matrix.app.name }}-vercel.json e2e-failures-${{ matrix.app.name }}-vercel.json + e2e-flaky-${{ matrix.app.name }}-vercel.json e2e-diagnostics-${{ matrix.app.name }}-vercel.json e2e-runtime-logs-${{ matrix.app.name }}-vercel.json retention-days: 7 @@ -892,7 +895,9 @@ jobs: uses: actions/upload-artifact@v4 with: name: e2e-results-local-dev-${{ matrix.app.name }}-${{ matrix.app.artifactSuffix }} - path: e2e-local-dev-${{ matrix.app.name }}-${{ matrix.app.artifactSuffix }}.json + path: | + e2e-local-dev-${{ matrix.app.name }}-${{ matrix.app.artifactSuffix }}.json + e2e-flaky-${{ matrix.app.name }}-local.json retention-days: 7 if-no-files-found: ignore @@ -981,7 +986,9 @@ jobs: uses: actions/upload-artifact@v4 with: name: e2e-results-local-prod-${{ matrix.app.name }}-${{ matrix.app.artifactSuffix }} - path: e2e-local-prod-${{ matrix.app.name }}-${{ matrix.app.artifactSuffix }}.json + path: | + e2e-local-prod-${{ matrix.app.name }}-${{ matrix.app.artifactSuffix }}.json + e2e-flaky-${{ matrix.app.name }}-local.json retention-days: 7 if-no-files-found: ignore @@ -1090,7 +1097,9 @@ jobs: uses: actions/upload-artifact@v4 with: name: e2e-results-local-postgres-${{ matrix.app.name }}-${{ matrix.app.artifactSuffix }} - path: e2e-local-postgres-${{ matrix.app.name }}-${{ matrix.app.artifactSuffix }}.json + path: | + e2e-local-postgres-${{ matrix.app.name }}-${{ matrix.app.artifactSuffix }}.json + e2e-flaky-${{ matrix.app.name }}-local.json retention-days: 7 if-no-files-found: ignore @@ -1244,7 +1253,9 @@ jobs: uses: actions/upload-artifact@v4 with: name: e2e-results-windows-nextjs-turbopack-${{ matrix.vm }} - path: e2e-windows-nextjs-turbopack-${{ matrix.vm }}.json + path: | + e2e-windows-nextjs-turbopack-${{ matrix.vm }}.json + e2e-flaky-nextjs-turbopack-local.json retention-days: 7 if-no-files-found: ignore diff --git a/packages/core/e2e/benchmark.test.ts b/packages/core/e2e/benchmark.test.ts index a64d95e717..addbf27bf8 100644 --- a/packages/core/e2e/benchmark.test.ts +++ b/packages/core/e2e/benchmark.test.ts @@ -752,7 +752,10 @@ function datadogRunSearchUrl(runId: string): string { return `https://app.datadoghq.com/apm/traces?query=${query}`; } -describe('workflow benchmarks', () => { +// A benchmark failure is a measurement, not a flake — opt out of the CI-wide +// e2e retry (vitest.config.ts) so a regression isn't papered over by a +// second, luckier sample (and scenario runtime isn't doubled). +describe('workflow benchmarks', { retry: 0 }, () => { // Preflight: prove the deployment executes workflows (and the trigger route // works) before any scenario spends its attempt budget. Without this, a // target that accepts run creation but never executes runs (e.g. queue not diff --git a/packages/core/e2e/event-log-race-repro.test.ts b/packages/core/e2e/event-log-race-repro.test.ts index b577df7b5c..002f2770cc 100644 --- a/packages/core/e2e/event-log-race-repro.test.ts +++ b/packages/core/e2e/event-log-race-repro.test.ts @@ -990,7 +990,10 @@ async function runScenario( // first and no summary is ever rendered. const testTimeoutMs = config.budgetMs + config.runTimeoutMs + 60_000; -describe('event log race repro', () => { +// This harness's failures ARE the signal it exists to produce, and a single +// pass runs for the whole configured budget — never let the CI-wide e2e +// retry (vitest.config.ts) re-run it. +describe('event log race repro', { retry: 0 }, () => { beforeAll(() => { setupWorld(deploymentUrl); diff --git a/packages/core/e2e/github-reporter.ts b/packages/core/e2e/github-reporter.ts index 4365690f43..dac7a44ea0 100644 --- a/packages/core/e2e/github-reporter.ts +++ b/packages/core/e2e/github-reporter.ts @@ -2,7 +2,10 @@ * Custom vitest reporter that emits GitHub Actions annotations for failed tests. * * When running in CI, failed e2e tests produce `::error` workflow commands that - * surface as annotations in the GitHub Actions UI and on PR file diffs. + * surface as annotations in the GitHub Actions UI and on PR file diffs. Tests + * that only passed after a retry (see `retry` in vitest.config.ts) produce + * `::warning` annotations and a `e2e-flaky-*.json` sidecar, so the retry that + * keeps a racy test from failing the job does not also hide the race. * * Also writes an enriched JSON sidecar file (`e2e-failures-*.json`) with * per-test failure details including run IDs and dashboard links, which the @@ -37,8 +40,16 @@ interface DiagnosticsEntry { timestamp: string; } +interface FlakyTestInfo { + testName: string; + fullName: string; + file: string; + retryCount: number; +} + export default class GithubAnnotationReporter implements Reporter { private failedTests: FailedTestInfo[] = []; + private flakyTests: FlakyTestInfo[] = []; onTestRunEnd(testModules: ReadonlyArray) { for (const module of testModules) { @@ -49,18 +60,36 @@ export default class GithubAnnotationReporter implements Reporter { // Enrich failures with diagnostics sidecar data (run IDs, dashboard URLs) this.enrichFromDiagnosticsSidecar(); this.writeFailuresSidecar(); + } - // Emit GitHub Actions annotations — this runs after vitest's own - // output is done, so ::error commands won't be mangled by ANSI codes. - if (process.env.CI) { - this.emitAnnotations(); - } + if (this.flakyTests.length > 0) { + this.writeFlakySidecar(); + } + + // Emit GitHub Actions annotations — this runs after vitest's own + // output is done, so ::error commands won't be mangled by ANSI codes. + if (process.env.CI) { + this.emitAnnotations(); } } private collectFailures(module: TestModule) { for (const test of module.children.allTests()) { const result = test.result(); + + if (result.state === 'passed') { + const retryCount = test.diagnostic()?.retryCount ?? 0; + if (retryCount > 0) { + this.flakyTests.push({ + testName: test.name, + fullName: test.fullName, + file: module.moduleId, + retryCount, + }); + } + continue; + } + if (result.state !== 'failed') continue; const errors = result.errors || []; @@ -127,6 +156,17 @@ export default class GithubAnnotationReporter implements Reporter { * rather than the workflow source file (which may be a symlink). */ private emitAnnotations() { + for (const test of this.flakyTests) { + const title = `E2E flaky: ${test.testName}`; + const body = `Passed only after ${test.retryCount} retr${ + test.retryCount === 1 ? 'y' : 'ies' + } — this test lost a race on its first attempt.`; + const relFile = path.relative(process.cwd(), test.file); + process.stdout.write( + `\n::warning file=${relFile},title=${title}::${body}\n` + ); + } + for (const test of this.failedTests) { const parts = [test.errorMessage.split('\n')[0].slice(0, 150)]; if (test.runId) parts.push(`Run: ${test.runId}`); @@ -155,4 +195,16 @@ export default class GithubAnnotationReporter implements Reporter { fs.writeFileSync(filePath, JSON.stringify(this.failedTests, null, 2)); } + + private writeFlakySidecar() { + const appName = process.env.APP_NAME || 'unknown'; + const isVercel = !!process.env.WORKFLOW_VERCEL_ENV; + const backend = isVercel ? 'vercel' : 'local'; + const filePath = path.resolve( + process.cwd(), + `e2e-flaky-${appName}-${backend}.json` + ); + + fs.writeFileSync(filePath, JSON.stringify(this.flakyTests, null, 2)); + } } diff --git a/vitest.config.ts b/vitest.config.ts index 4addd1e955..ad59d47870 100644 --- a/vitest.config.ts +++ b/vitest.config.ts @@ -3,6 +3,15 @@ import { configDefaults, defineConfig } from 'vitest/config'; export default defineConfig({ test: { testTimeout: 60_000, + // The e2e suites drive real deployments, so individual tests can lose + // timing races (queue delays, cold starts, watcher latency) that a + // second attempt absorbs. One CI retry keeps a single racy test from + // failing a 20+ minute matrix job; retried tests stay visible — the + // github-reporter annotates them and the PR comment lists them — so + // real races still get looked at. Harnesses where a failure is itself + // the signal (event-log-race-repro, benchmarks) pin `retry: 0` locally. + // Local runs keep retry at 0 so races reproduce while debugging. + retry: process.env.CI ? 1 : 0, // Positional file arguments are regex filters, not paths, so // `vitest run packages/core/e2e/x.test.ts` also matches // `.claude/worktrees//packages/core/e2e/x.test.ts` when agent