From 4b7af1be41d79f0ee00a1e8d5cdc420931cc7d92 Mon Sep 17 00:00:00 2001 From: Dotta Date: Thu, 3 Sep 2026 09:52:34 -0500 Subject: [PATCH] test(runner): retire unqualified DeepSeek plan cell --- tests/runner-e2e/README.md | 13 ++++--- tests/runner-e2e/catalog.test.ts | 26 +++++++++---- tests/runner-e2e/catalog.ts | 63 ++++++++++++++++++-------------- tests/runner-e2e/history.test.ts | 4 +- tests/runner-e2e/types.ts | 1 + 5 files changed, 65 insertions(+), 42 deletions(-) diff --git a/tests/runner-e2e/README.md b/tests/runner-e2e/README.md index e14c66c77d..58ca40b76a 100644 --- a/tests/runner-e2e/README.md +++ b/tests/runner-e2e/README.md @@ -82,9 +82,12 @@ workflows: 42 cells. Its cases are: `openrouter-model-breadth` (**OpenRouter Model Breadth**) is four qualified models from the tracked weekly tool-capable ranking snapshot × native OpenCode -× local × three workflows: 12 cells. Xiaomi MiMo V2.5 remains recorded in the -immutable ranking snapshot but is excluded from paid qualification because its -latency repeatedly exhausts the cell deadline. Its cases are: +× local, with 11 supported model/workflow cells. Xiaomi MiMo V2.5 remains +recorded in the immutable ranking snapshot but is excluded from paid +qualification because its latency repeatedly exhausts the cell deadline. +DeepSeek V4 Flash remains qualified for hello and question/resume, but its Plan +cell is excluded after three successful semantic completions consistently +ignored the required exact final response. Its cases are: - `hello-complete`: a basic nonce response and explicit Done transition; - `question-resume-complete`: one structured question, browser selection of @@ -100,7 +103,7 @@ duplicating the final response. The second workflow restarts the isolated Paperclip server while the interaction is waiting, reloads that state, and then resumes it. The suite has no Daytona cells. -The complete catalog is 68 cells and 118 expected paid agent turns. Follow-up +The complete catalog is 67 cells and 116 expected paid agent turns. Follow-up steps remain ordered within their cell; all other cells are independent. Narrow selectors are strongly recommended while developing fixtures. @@ -292,7 +295,7 @@ Set `RUNNER_E2E_AWS_ENABLED=true` to route paid cells to the repository-scoped ephemeral AWS RunsOn fleet selected by `runs-on/fleet=paperclip-public-pr-x64/env=public-ci`. Any other value uses the proven GitHub-hosted `ubuntu-latest` target. Set `RUNNER_E2E_MAX_PARALLEL` to an -integer from 1–100 on AWS (default 100); use at least 68 to run the current +integer from 1–100 on AWS (default 100); use at least 67 to run the current complete catalog in one wave. The fallback runner retains its 1–57 limit and default of 32. Multi-turn steps are sequential inside their cell while independent cells overlap. Artifacts and merged HTML/JUnit/normalized reports diff --git a/tests/runner-e2e/catalog.test.ts b/tests/runner-e2e/catalog.test.ts index 4116f8ec1c..6c820b8523 100644 --- a/tests/runner-e2e/catalog.test.ts +++ b/tests/runner-e2e/catalog.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "vitest"; import { runnerEnvironments, runnerMatrix, + openRouterBreadthExcludedExecutionIds, openRouterBreadthExcludedModelIds, openRouterBreadthProfiles, openRouterBreadthTasks, @@ -28,10 +29,10 @@ describe("runner E2E catalog", () => { expect(localIntegrityTasks).toHaveLength(2); expect(openRouterBreadthTasks).toHaveLength(3); expect(runnerSuites.map((suite) => suite.expectedMatrixSize)).toEqual([ - 42, 14, 12, + 42, 14, 11, ]); - expect(validateRunnerCatalog()).toHaveLength(68); - expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(68); + expect(validateRunnerCatalog()).toHaveLength(67); + expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(67); expect( runnerMatrix.filter((entry) => entry.suite.id === "core-compatibility"), ).toHaveLength(42); @@ -44,17 +45,26 @@ describe("runner E2E catalog", () => { runnerMatrix.filter( (entry) => entry.suite.id === "openrouter-model-breadth", ), - ).toHaveLength(12); + ).toHaveLength(11); expect( runnerMatrix.reduce( (total, execution) => total + execution.task.expectedRunCount, 0, ), - ).toBe(118); + ).toBe(116); }); it("derives the qualified local native OpenCode profiles from the ranked snapshot", () => { expect(openRouterBreadthExcludedModelIds).toEqual(["xiaomi/mimo-v2.5"]); + expect(openRouterBreadthExcludedExecutionIds).toEqual([ + "openrouter-model-breadth.openrouter-deepseek-deepseek-v4-flash-0731.local.plan-approve-complete", + ]); + expect( + runnerMatrix.some( + (execution) => + execution.id === openRouterBreadthExcludedExecutionIds[0], + ), + ).toBe(false); expect( openRouterBreadthProfiles.map((profile) => profile.ranking?.rank), ).toEqual([1, 3, 4, 5]); @@ -313,7 +323,7 @@ describe("runner E2E selectors", () => { const selected = selectRunnerExecutions( parseRunnerSelectors(["--suite", "openrouter-model-breadth"]), ); - expect(selected).toHaveLength(12); + expect(selected).toHaveLength(11); expect( selected.every( (entry) => @@ -350,9 +360,9 @@ describe("runner E2E selectors", () => { const jobs = buildMatrixJobs( selectRunnerExecutions(parseRunnerSelectors(["--all"])), ); - expect(jobs).toHaveLength(68); + expect(jobs).toHaveLength(67); expect(jobs.filter((job) => job.needsDaytona)).toHaveLength(21); - expect(new Set(jobs.map((job) => job.executionId)).size).toBe(68); + expect(new Set(jobs.map((job) => job.executionId)).size).toBe(67); expect( jobs.every((job) => runnerMatrix.some( diff --git a/tests/runner-e2e/catalog.ts b/tests/runner-e2e/catalog.ts index 5e7039b895..531a35befb 100644 --- a/tests/runner-e2e/catalog.ts +++ b/tests/runner-e2e/catalog.ts @@ -263,6 +263,9 @@ export const runnerProfiles: readonly RunnerProfileFixture[] = [ ] as const; export const openRouterBreadthExcludedModelIds = ["xiaomi/mimo-v2.5"] as const; +export const openRouterBreadthExcludedExecutionIds = [ + "openrouter-model-breadth.openrouter-deepseek-deepseek-v4-flash-0731.local.plan-approve-complete", +] as const; const openRouterBreadthExcludedModelIdSet = new Set( openRouterBreadthExcludedModelIds, ); @@ -743,13 +746,15 @@ export const runnerSuites: readonly RunnerSuiteFixture[] = [ profiles: openRouterBreadthProfiles, environments: [localEnvironment], tasks: openRouterBreadthTasks, - expectedMatrixSize: 12, + excludedExecutionIds: openRouterBreadthExcludedExecutionIds, + expectedMatrixSize: 11, definitionMetadata: { rankingSnapshotId: openRouterRankingSnapshot.snapshotId, rankingContentHash: openRouterRankingSnapshot.contentHash, rankingCapturedAt: openRouterRankingSnapshot.capturedAt, rankingSourceUrl: openRouterRankingSnapshot.sourceUrl, excludedModelIds: openRouterBreadthExcludedModelIds, + excludedExecutionIds: openRouterBreadthExcludedExecutionIds, }, }, ] as const; @@ -772,6 +777,7 @@ export function suiteDefinitionHash(suite: RunnerSuiteFixture) { restartServerBeforeQuestionAnswer: task.restartServerBeforeQuestionAnswer ?? false, })), + excludedExecutionIds: [...(suite.excludedExecutionIds ?? [])].sort(), metadata: suite.definitionMetadata ?? null, }), ) @@ -781,36 +787,39 @@ export function suiteDefinitionHash(suite: RunnerSuiteFixture) { export function buildRunnerMatrix( suites: readonly RunnerSuiteFixture[] = runnerSuites, ): MatrixExecution[] { - return suites.flatMap((suite) => - suite.profiles.flatMap((profile) => + return suites.flatMap((suite) => { + const excludedExecutionIds = new Set(suite.excludedExecutionIds ?? []); + return suite.profiles.flatMap((profile) => suite.environments .filter((environment) => profile.supportedEnvironments.includes(environment.id), ) .flatMap((environment) => - suite.tasks.map((task) => ({ - id: `${suite.id}.${profile.id}.${environment.id}.${task.id}`, - suite, - suiteDefinitionHash: suiteDefinitionHash(suite), - profile, - environment, - task, - groups: [ - ...new Set([ - ...suite.groups, - ...profile.groups, - ...environment.groups, - ...task.groups, - ]), - ], - requiredCredentials: [ - profile.credential, - ...(environment.credential ? [environment.credential] : []), - ], - })), + suite.tasks + .map((task) => ({ + id: `${suite.id}.${profile.id}.${environment.id}.${task.id}`, + suite, + suiteDefinitionHash: suiteDefinitionHash(suite), + profile, + environment, + task, + groups: [ + ...new Set([ + ...suite.groups, + ...profile.groups, + ...environment.groups, + ...task.groups, + ]), + ], + requiredCredentials: [ + profile.credential, + ...(environment.credential ? [environment.credential] : []), + ], + })) + .filter((execution) => !excludedExecutionIds.has(execution.id)), ), - ), - ); + ); + }); } function duplicateIds(values: readonly { id: string }[]) { @@ -954,8 +963,8 @@ export function validateRunnerCatalog(): MatrixExecution[] { ); } } - if (matrix.length !== 68) - throw new Error(`Expected 68 runner executions; received ${matrix.length}`); + if (matrix.length !== 67) + throw new Error(`Expected 67 runner executions; received ${matrix.length}`); return matrix; } diff --git a/tests/runner-e2e/history.test.ts b/tests/runner-e2e/history.test.ts index 0e2941da3e..0ae98fb272 100644 --- a/tests/runner-e2e/history.test.ts +++ b/tests/runner-e2e/history.test.ts @@ -151,8 +151,8 @@ describe("runner E2E campaign history", () => { expect(index).toContain("Runner E2E campaigns"); expect(index).toContain("complete-green"); expect(index).toContain("complete-red"); - expect(index).toContain("68/68 passed"); - expect(index).toContain("67/68 passed"); + expect(index).toContain("67/67 passed"); + expect(index).toContain("66/67 passed"); expect(index).toContain("Open report →"); expect(index).toContain( "Visual evidence remains in access-controlled workflow artifacts", diff --git a/tests/runner-e2e/types.ts b/tests/runner-e2e/types.ts index eeafacf0f2..3191832682 100644 --- a/tests/runner-e2e/types.ts +++ b/tests/runner-e2e/types.ts @@ -156,6 +156,7 @@ export interface RunnerSuiteFixture { profiles: readonly RunnerProfileFixture[]; environments: readonly EnvironmentFixture[]; tasks: readonly RunnerTaskFixture[]; + excludedExecutionIds?: readonly string[]; expectedMatrixSize: number; definitionMetadata?: Readonly>; }