From e228846d4a2929e21dbcdbc4637672e162b114b1 Mon Sep 17 00:00:00 2001 From: Dotta Date: Thu, 3 Sep 2026 08:49:05 -0500 Subject: [PATCH] test(runner): retire Xiaomi breadth cells --- tests/runner-e2e/README.md | 12 ++++--- tests/runner-e2e/catalog.test.ts | 24 +++++++------ tests/runner-e2e/catalog.ts | 58 +++++++++++++++++++------------- tests/runner-e2e/history.test.ts | 12 +++---- 4 files changed, 60 insertions(+), 46 deletions(-) diff --git a/tests/runner-e2e/README.md b/tests/runner-e2e/README.md index b66c7da8fe..e14c66c77d 100644 --- a/tests/runner-e2e/README.md +++ b/tests/runner-e2e/README.md @@ -80,9 +80,11 @@ workflows: 42 cells. Its cases are: same Plan, browser acceptance of the new revision, and verified execution; - `ask-question`: a direct answer from a task created in Ask mode. -`openrouter-model-breadth` (**OpenRouter Model Breadth**) is five models from -the tracked weekly tool-capable ranking snapshot × native OpenCode × local × -three workflows: 15 cells. Its cases are: +`openrouter-model-breadth` (**OpenRouter Model Breadth**) is four qualified +models from the tracked weekly tool-capable ranking snapshot × native OpenCode +× local × three workflows: 12 cells. Xiaomi MiMo V2.5 remains recorded in the +immutable ranking snapshot but is excluded from paid qualification because its +latency repeatedly exhausts the cell deadline. Its cases are: - `hello-complete`: a basic nonce response and explicit Done transition; - `question-resume-complete`: one structured question, browser selection of @@ -98,7 +100,7 @@ duplicating the final response. The second workflow restarts the isolated Paperclip server while the interaction is waiting, reloads that state, and then resumes it. The suite has no Daytona cells. -The complete catalog is 71 cells and 123 expected paid agent turns. Follow-up +The complete catalog is 68 cells and 118 expected paid agent turns. Follow-up steps remain ordered within their cell; all other cells are independent. Narrow selectors are strongly recommended while developing fixtures. @@ -290,7 +292,7 @@ Set `RUNNER_E2E_AWS_ENABLED=true` to route paid cells to the repository-scoped ephemeral AWS RunsOn fleet selected by `runs-on/fleet=paperclip-public-pr-x64/env=public-ci`. Any other value uses the proven GitHub-hosted `ubuntu-latest` target. Set `RUNNER_E2E_MAX_PARALLEL` to an -integer from 1–100 on AWS (default 100); use at least 71 to run the current +integer from 1–100 on AWS (default 100); use at least 68 to run the current complete catalog in one wave. The fallback runner retains its 1–57 limit and default of 32. Multi-turn steps are sequential inside their cell while independent cells overlap. Artifacts and merged HTML/JUnit/normalized reports diff --git a/tests/runner-e2e/catalog.test.ts b/tests/runner-e2e/catalog.test.ts index c81914b7c2..6dbcb23090 100644 --- a/tests/runner-e2e/catalog.test.ts +++ b/tests/runner-e2e/catalog.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "vitest"; import { runnerEnvironments, runnerMatrix, + openRouterBreadthExcludedModelIds, openRouterBreadthProfiles, openRouterBreadthTasks, localIntegrityTasks, @@ -21,16 +22,16 @@ import { describe("runner E2E catalog", () => { it("validates the core, local-integrity, and breadth suites", () => { expect(runnerProfiles).toHaveLength(7); - expect(openRouterBreadthProfiles).toHaveLength(5); + expect(openRouterBreadthProfiles).toHaveLength(4); expect(runnerEnvironments).toHaveLength(2); expect(runnerTasks).toHaveLength(3); expect(localIntegrityTasks).toHaveLength(2); expect(openRouterBreadthTasks).toHaveLength(3); expect(runnerSuites.map((suite) => suite.expectedMatrixSize)).toEqual([ - 42, 14, 15, + 42, 14, 12, ]); - expect(validateRunnerCatalog()).toHaveLength(71); - expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(71); + expect(validateRunnerCatalog()).toHaveLength(68); + expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(68); expect( runnerMatrix.filter((entry) => entry.suite.id === "core-compatibility"), ).toHaveLength(42); @@ -43,19 +44,20 @@ describe("runner E2E catalog", () => { runnerMatrix.filter( (entry) => entry.suite.id === "openrouter-model-breadth", ), - ).toHaveLength(15); + ).toHaveLength(12); expect( runnerMatrix.reduce( (total, execution) => total + execution.task.expectedRunCount, 0, ), - ).toBe(123); + ).toBe(118); }); - it("derives five local native OpenCode profiles from the ranked snapshot", () => { + it("derives the qualified local native OpenCode profiles from the ranked snapshot", () => { + expect(openRouterBreadthExcludedModelIds).toEqual(["xiaomi/mimo-v2.5"]); expect( openRouterBreadthProfiles.map((profile) => profile.ranking?.rank), - ).toEqual([1, 2, 3, 4, 5]); + ).toEqual([1, 3, 4, 5]); expect( openRouterBreadthProfiles.every( (profile) => @@ -313,7 +315,7 @@ describe("runner E2E selectors", () => { const selected = selectRunnerExecutions( parseRunnerSelectors(["--suite", "openrouter-model-breadth"]), ); - expect(selected).toHaveLength(15); + expect(selected).toHaveLength(12); expect( selected.every( (entry) => @@ -350,9 +352,9 @@ describe("runner E2E selectors", () => { const jobs = buildMatrixJobs( selectRunnerExecutions(parseRunnerSelectors(["--all"])), ); - expect(jobs).toHaveLength(71); + expect(jobs).toHaveLength(68); expect(jobs.filter((job) => job.needsDaytona)).toHaveLength(21); - expect(new Set(jobs.map((job) => job.executionId)).size).toBe(71); + expect(new Set(jobs.map((job) => job.executionId)).size).toBe(68); expect( jobs.every((job) => runnerMatrix.some( diff --git a/tests/runner-e2e/catalog.ts b/tests/runner-e2e/catalog.ts index 53c859c553..492d4e2310 100644 --- a/tests/runner-e2e/catalog.ts +++ b/tests/runner-e2e/catalog.ts @@ -262,28 +262,37 @@ export const runnerProfiles: readonly RunnerProfileFixture[] = [ }), ] as const; +export const openRouterBreadthExcludedModelIds = ["xiaomi/mimo-v2.5"] as const; +const openRouterBreadthExcludedModelIdSet = new Set( + openRouterBreadthExcludedModelIds, +); + export const openRouterBreadthProfiles: readonly RunnerProfileFixture[] = - openRouterRankingSnapshot.models.map((rankedModel) => - nativeProfile({ - id: openRouterProfileId(rankedModel.id), - label: `#${rankedModel.rank} ${rankedModel.name}`, - provider: "opencode", - model: `openrouter/${rankedModel.id}`, - credential: "OPENROUTER_API_KEY", - supportedEnvironments: ["local"], - modelQualification: { - source: "openrouter_rankings_snapshot", - qualificationId: `${openRouterRankingSnapshot.snapshotId}:${rankedModel.rank}`, - }, - ranking: { - rank: rankedModel.rank, - canonicalModelId: rankedModel.id, - snapshotId: openRouterRankingSnapshot.snapshotId, - capturedAt: openRouterRankingSnapshot.capturedAt, - sourceUrl: openRouterRankingSnapshot.sourceUrl, - }, - }), - ); + openRouterRankingSnapshot.models + .filter( + (rankedModel) => !openRouterBreadthExcludedModelIdSet.has(rankedModel.id), + ) + .map((rankedModel) => + nativeProfile({ + id: openRouterProfileId(rankedModel.id), + label: `#${rankedModel.rank} ${rankedModel.name}`, + provider: "opencode", + model: `openrouter/${rankedModel.id}`, + credential: "OPENROUTER_API_KEY", + supportedEnvironments: ["local"], + modelQualification: { + source: "openrouter_rankings_snapshot", + qualificationId: `${openRouterRankingSnapshot.snapshotId}:${rankedModel.rank}`, + }, + ranking: { + rank: rankedModel.rank, + canonicalModelId: rankedModel.id, + snapshotId: openRouterRankingSnapshot.snapshotId, + capturedAt: openRouterRankingSnapshot.capturedAt, + sourceUrl: openRouterRankingSnapshot.sourceUrl, + }, + }), + ); function requiredDaytonaSecret(input: EnvironmentFixtureBuildInput) { const apiKey = input.secretRefs.DAYTONA_API_KEY; @@ -734,12 +743,13 @@ export const runnerSuites: readonly RunnerSuiteFixture[] = [ profiles: openRouterBreadthProfiles, environments: [localEnvironment], tasks: openRouterBreadthTasks, - expectedMatrixSize: 15, + expectedMatrixSize: 12, definitionMetadata: { rankingSnapshotId: openRouterRankingSnapshot.snapshotId, rankingContentHash: openRouterRankingSnapshot.contentHash, rankingCapturedAt: openRouterRankingSnapshot.capturedAt, rankingSourceUrl: openRouterRankingSnapshot.sourceUrl, + excludedModelIds: openRouterBreadthExcludedModelIds, }, }, ] as const; @@ -944,8 +954,8 @@ export function validateRunnerCatalog(): MatrixExecution[] { ); } } - if (matrix.length !== 71) - throw new Error(`Expected 71 runner executions; received ${matrix.length}`); + if (matrix.length !== 68) + throw new Error(`Expected 68 runner executions; received ${matrix.length}`); return matrix; } diff --git a/tests/runner-e2e/history.test.ts b/tests/runner-e2e/history.test.ts index 6748b76ae5..0e2941da3e 100644 --- a/tests/runner-e2e/history.test.ts +++ b/tests/runner-e2e/history.test.ts @@ -68,15 +68,15 @@ describe("runner E2E campaign history", () => { expected: breadth.map((execution) => execution.id), results: breadth.map((execution) => result(execution, "passed")), }); - expect(campaign).toMatchObject({ complete: false, passed: 15, failed: 0 }); + expect(campaign).toMatchObject({ complete: false, passed: 12, failed: 0 }); expect(campaign.suites[0]).toMatchObject({ suiteId: "openrouter-model-breadth", complete: true, - selected: 15, + selected: 12, }); expect(campaign.billing).toMatchObject({ - reportedLlmCostUsd: 0.15, - llm: { inputTokens: 1_500, outputTokens: 375 }, + reportedLlmCostUsd: 0.12, + llm: { inputTokens: 1_200, outputTokens: 300 }, }); const history = mergeRunnerHistory( emptyRunnerHistory(), @@ -151,8 +151,8 @@ describe("runner E2E campaign history", () => { expect(index).toContain("Runner E2E campaigns"); expect(index).toContain("complete-green"); expect(index).toContain("complete-red"); - expect(index).toContain("71/71 passed"); - expect(index).toContain("70/71 passed"); + expect(index).toContain("68/68 passed"); + expect(index).toContain("67/68 passed"); expect(index).toContain("Open report →"); expect(index).toContain( "Visual evidence remains in access-controlled workflow artifacts",