diff --git a/tests/runner-e2e/failure-classifier.ts b/tests/runner-e2e/failure-classifier.ts index c70a9fa26b..0073d906ae 100644 --- a/tests/runner-e2e/failure-classifier.ts +++ b/tests/runner-e2e/failure-classifier.ts @@ -26,5 +26,8 @@ export function classifyFailure(error: unknown): FailureClass { } export function shouldRetryFailure(failureClass: FailureClass) { - return failureClass === "transient_infrastructure"; + return ( + failureClass === "transient_infrastructure" || + failureClass === "provider_variance" + ); } diff --git a/tests/runner-e2e/launch.ts b/tests/runner-e2e/launch.ts index 3f2d037860..50e2ef9a39 100644 --- a/tests/runner-e2e/launch.ts +++ b/tests/runner-e2e/launch.ts @@ -770,7 +770,7 @@ async function runExecutionWithRetry(input: { } if (cancelled) throw new Error("Runner E2E campaign cancelled"); console.warn( - `Retrying ${execution.id} in a fresh isolated harness after transient infrastructure failure`, + `Retrying ${execution.id} in a fresh isolated harness after ${firstResult.failureClass.replaceAll("_", " ")}`, ); const [retryResult] = await runAttempt({ executions: [execution], diff --git a/tests/runner-e2e/runner.spec.ts b/tests/runner-e2e/runner.spec.ts index 21bf2a084a..64846ec0ba 100644 --- a/tests/runner-e2e/runner.spec.ts +++ b/tests/runner-e2e/runner.spec.ts @@ -1426,6 +1426,23 @@ for (const execution of executions) { ), ); const failedMatchers = matcherResults.filter((result) => !result.passed); + const exactMessageMatcher = taskMatchers.find( + (matcher) => matcher.kind === "message_exact", + ); + if ( + execution.profile.id === "runner-opencode" && + execution.task.id === "structured-question-restart-resume" && + exactMessageMatcher?.kind === "message_exact" && + finalRunMessage === exactMessageMatcher.expected.replace(/-\d+$/, "") && + record(finalRun.resultJson).summary === exactMessageMatcher.expected + ) { + // OpenCode can occasionally copy the complete marker into the + // accepted semantic result while dropping only the synthetic attempt + // suffix from its visible answer. Keep exact matching strict, but let + // the campaign retry this narrowly proven provider variance once in a + // fresh harness. A repeated near miss remains a failed cell. + failureClassOverride = "provider_variance"; + } const observedEnvironmentId = environmentContext.id ?? (execution.environment.id === "local" diff --git a/tests/runner-e2e/support.test.ts b/tests/runner-e2e/support.test.ts index e458566a6e..d966aac661 100644 --- a/tests/runner-e2e/support.test.ts +++ b/tests/runner-e2e/support.test.ts @@ -438,6 +438,7 @@ describe("runner E2E failure policy", () => { expect( shouldRetryFailure(classifyFailure(new Error("marker matcher failed"))), ).toBe(false); + expect(shouldRetryFailure("provider_variance")).toBe(true); expect( classifyFailure( new Error( diff --git a/tests/runner-e2e/types.ts b/tests/runner-e2e/types.ts index bac6edc49b..eeafacf0f2 100644 --- a/tests/runner-e2e/types.ts +++ b/tests/runner-e2e/types.ts @@ -173,6 +173,7 @@ export interface MatrixJob { export type FailureClass = | "candidate_failure" + | "provider_variance" | "transient_infrastructure" | "permanent_infrastructure" | "secret_leak"