diff --git a/tests/runner-e2e/README.md b/tests/runner-e2e/README.md index be091a341a..ad62738f60 100644 --- a/tests/runner-e2e/README.md +++ b/tests/runner-e2e/README.md @@ -82,12 +82,16 @@ workflows: 42 cells. Its cases are: `openrouter-model-breadth` (**OpenRouter Model Breadth**) is four qualified models from the tracked weekly tool-capable ranking snapshot × native OpenCode -× local, with 11 supported model/workflow cells. Xiaomi MiMo V2.5 remains +× local, with 10 supported model/workflow cells. Xiaomi MiMo V2.5 remains recorded in the immutable ranking snapshot but is excluded from paid qualification because its latency repeatedly exhausts the cell deadline. DeepSeek V4 Flash remains qualified for hello and question/resume, but its Plan cell is excluded after three successful semantic completions consistently -ignored the required exact final response. Its cases are: +ignored the required exact final response. Tencent HY3 likewise remains +qualified for hello and question/resume, but its Plan cell is excluded after +two fresh attempts completed every durable Plan and finalization operation yet +consistently replaced the required exact visible terminal marker with prose. +Its cases are: - `hello-complete`: a basic nonce response and explicit Done transition; - `question-resume-complete`: one structured question, browser selection of @@ -103,9 +107,10 @@ duplicating the final response. The second workflow restarts the isolated Paperclip server while the interaction is waiting, reloads that state, and then resumes it. The suite has no Daytona cells. -The complete catalog is 67 cells and 116 expected paid agent turns. Follow-up -steps remain ordered within their cell; all other cells are independent. -Narrow selectors are strongly recommended while developing fixtures. +The complete catalog is 66 cells (45 local and 21 Daytona) and 114 expected +paid agent turns. Follow-up steps remain ordered within their cell; all other +cells are independent. Narrow selectors are strongly recommended while +developing fixtures. `--suite`, `--group`, `--profile`, `--environment`, and `--case` are repeatable. Repeated values in one dimension use OR semantics; dimensions and repeated groups use @@ -330,7 +335,7 @@ Set `RUNNER_E2E_AWS_ENABLED=true` to route paid cells to the repository-scoped ephemeral AWS RunsOn fleet selected by `runs-on/fleet=paperclip-public-pr-x64/env=public-ci`. Any other value uses the proven GitHub-hosted `ubuntu-latest` target. Set `RUNNER_E2E_MAX_PARALLEL` to an -integer from 1–100 on AWS (default 100); use at least 67 to run the current +integer from 1–100 on AWS (default 100); use at least 66 to run the current complete catalog in one wave. The fallback runner retains its 1–57 limit and default of 32. Multi-turn steps are sequential inside their cell while independent cells overlap. Artifacts and merged HTML/JUnit/normalized reports diff --git a/tests/runner-e2e/catalog.test.ts b/tests/runner-e2e/catalog.test.ts index 6c820b8523..409c10cd84 100644 --- a/tests/runner-e2e/catalog.test.ts +++ b/tests/runner-e2e/catalog.test.ts @@ -29,10 +29,10 @@ describe("runner E2E catalog", () => { expect(localIntegrityTasks).toHaveLength(2); expect(openRouterBreadthTasks).toHaveLength(3); expect(runnerSuites.map((suite) => suite.expectedMatrixSize)).toEqual([ - 42, 14, 11, + 42, 14, 10, ]); - expect(validateRunnerCatalog()).toHaveLength(67); - expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(67); + expect(validateRunnerCatalog()).toHaveLength(66); + expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(66); expect( runnerMatrix.filter((entry) => entry.suite.id === "core-compatibility"), ).toHaveLength(42); @@ -45,26 +45,36 @@ describe("runner E2E catalog", () => { runnerMatrix.filter( (entry) => entry.suite.id === "openrouter-model-breadth", ), - ).toHaveLength(11); + ).toHaveLength(10); expect( runnerMatrix.reduce( (total, execution) => total + execution.task.expectedRunCount, 0, ), - ).toBe(116); + ).toBe(114); }); it("derives the qualified local native OpenCode profiles from the ranked snapshot", () => { expect(openRouterBreadthExcludedModelIds).toEqual(["xiaomi/mimo-v2.5"]); expect(openRouterBreadthExcludedExecutionIds).toEqual([ "openrouter-model-breadth.openrouter-deepseek-deepseek-v4-flash-0731.local.plan-approve-complete", + "openrouter-model-breadth.openrouter-tencent-hy3.local.plan-approve-complete", ]); expect( - runnerMatrix.some( - (execution) => - execution.id === openRouterBreadthExcludedExecutionIds[0], + openRouterBreadthExcludedExecutionIds.every( + (excludedExecutionId) => + !runnerMatrix.some( + (execution) => execution.id === excludedExecutionId, + ), ), - ).toBe(false); + ).toBe(true); + expect( + runnerMatrix + .filter( + (execution) => execution.profile.id === "openrouter-tencent-hy3", + ) + .map((execution) => execution.task.id), + ).toEqual(["hello-complete", "question-resume-complete"]); expect( openRouterBreadthProfiles.map((profile) => profile.ranking?.rank), ).toEqual([1, 3, 4, 5]); @@ -323,7 +333,7 @@ describe("runner E2E selectors", () => { const selected = selectRunnerExecutions( parseRunnerSelectors(["--suite", "openrouter-model-breadth"]), ); - expect(selected).toHaveLength(11); + expect(selected).toHaveLength(10); expect( selected.every( (entry) => @@ -360,9 +370,10 @@ describe("runner E2E selectors", () => { const jobs = buildMatrixJobs( selectRunnerExecutions(parseRunnerSelectors(["--all"])), ); - expect(jobs).toHaveLength(67); + expect(jobs).toHaveLength(66); expect(jobs.filter((job) => job.needsDaytona)).toHaveLength(21); - expect(new Set(jobs.map((job) => job.executionId)).size).toBe(67); + expect(jobs.filter((job) => !job.needsDaytona)).toHaveLength(45); + expect(new Set(jobs.map((job) => job.executionId)).size).toBe(66); expect( jobs.every((job) => runnerMatrix.some( diff --git a/tests/runner-e2e/catalog.ts b/tests/runner-e2e/catalog.ts index 531a35befb..204d97d87c 100644 --- a/tests/runner-e2e/catalog.ts +++ b/tests/runner-e2e/catalog.ts @@ -265,6 +265,7 @@ export const runnerProfiles: readonly RunnerProfileFixture[] = [ export const openRouterBreadthExcludedModelIds = ["xiaomi/mimo-v2.5"] as const; export const openRouterBreadthExcludedExecutionIds = [ "openrouter-model-breadth.openrouter-deepseek-deepseek-v4-flash-0731.local.plan-approve-complete", + "openrouter-model-breadth.openrouter-tencent-hy3.local.plan-approve-complete", ] as const; const openRouterBreadthExcludedModelIdSet = new Set( openRouterBreadthExcludedModelIds, @@ -747,7 +748,7 @@ export const runnerSuites: readonly RunnerSuiteFixture[] = [ environments: [localEnvironment], tasks: openRouterBreadthTasks, excludedExecutionIds: openRouterBreadthExcludedExecutionIds, - expectedMatrixSize: 11, + expectedMatrixSize: 10, definitionMetadata: { rankingSnapshotId: openRouterRankingSnapshot.snapshotId, rankingContentHash: openRouterRankingSnapshot.contentHash, @@ -963,8 +964,8 @@ export function validateRunnerCatalog(): MatrixExecution[] { ); } } - if (matrix.length !== 67) - throw new Error(`Expected 67 runner executions; received ${matrix.length}`); + if (matrix.length !== 66) + throw new Error(`Expected 66 runner executions; received ${matrix.length}`); return matrix; } diff --git a/tests/runner-e2e/history.test.ts b/tests/runner-e2e/history.test.ts index 8a6cec4c7c..2a0c85c37a 100644 --- a/tests/runner-e2e/history.test.ts +++ b/tests/runner-e2e/history.test.ts @@ -68,16 +68,16 @@ describe("runner E2E campaign history", () => { expected: breadth.map((execution) => execution.id), results: breadth.map((execution) => result(execution, "passed")), }); - expect(campaign).toMatchObject({ complete: false, passed: 11, failed: 0 }); + expect(campaign).toMatchObject({ complete: false, passed: 10, failed: 0 }); expect(campaign.suites[0]).toMatchObject({ suiteId: "openrouter-model-breadth", complete: true, - selected: 11, + selected: 10, }); expect(campaign.billing).toMatchObject({ - llm: { inputTokens: 1_100, outputTokens: 275 }, + llm: { inputTokens: 1_000, outputTokens: 250 }, }); - expect(campaign.billing.reportedLlmCostUsd).toBeCloseTo(0.11, 10); + expect(campaign.billing.reportedLlmCostUsd).toBeCloseTo(0.1, 10); const history = mergeRunnerHistory( emptyRunnerHistory(), campaignHistoryRecord(campaign, "https://history.example/runner-e2e"), @@ -151,8 +151,8 @@ describe("runner E2E campaign history", () => { expect(index).toContain("Runner E2E campaigns"); expect(index).toContain("complete-green"); expect(index).toContain("complete-red"); - expect(index).toContain("67/67 passed"); - expect(index).toContain("66/67 passed"); + expect(index).toContain("66/66 passed"); + expect(index).toContain("65/66 passed"); expect(index).toContain("Open report →"); expect(index).toContain( "Visual evidence remains in access-controlled workflow artifacts",