test(runner): retire Xiaomi breadth cells

This commit is contained in:
Dotta 2026-09-03 08:49:05 -05:00
parent 05c7f3a857
commit e228846d4a
4 changed files with 60 additions and 46 deletions

View File

@ -80,9 +80,11 @@ workflows: 42 cells. Its cases are:
same Plan, browser acceptance of the new revision, and verified execution;
- `ask-question`: a direct answer from a task created in Ask mode.
`openrouter-model-breadth` (**OpenRouter Model Breadth**) is five models from
the tracked weekly tool-capable ranking snapshot × native OpenCode × local ×
three workflows: 15 cells. Its cases are:
`openrouter-model-breadth` (**OpenRouter Model Breadth**) is four qualified
models from the tracked weekly tool-capable ranking snapshot × native OpenCode
× local × three workflows: 12 cells. Xiaomi MiMo V2.5 remains recorded in the
immutable ranking snapshot but is excluded from paid qualification because its
latency repeatedly exhausts the cell deadline. Its cases are:
- `hello-complete`: a basic nonce response and explicit Done transition;
- `question-resume-complete`: one structured question, browser selection of
@ -98,7 +100,7 @@ duplicating the final response. The second workflow restarts the isolated
Paperclip server while the interaction is waiting, reloads that state, and
then resumes it. The suite has no Daytona cells.
The complete catalog is 71 cells and 123 expected paid agent turns. Follow-up
The complete catalog is 68 cells and 118 expected paid agent turns. Follow-up
steps remain ordered within their cell; all other cells are independent.
Narrow selectors are strongly recommended while developing fixtures.
@ -290,7 +292,7 @@ Set `RUNNER_E2E_AWS_ENABLED=true` to route paid cells to the repository-scoped
ephemeral AWS RunsOn fleet selected by
`runs-on/fleet=paperclip-public-pr-x64/env=public-ci`. Any other value uses the
proven GitHub-hosted `ubuntu-latest` target. Set `RUNNER_E2E_MAX_PARALLEL` to an
integer from 1–100 on AWS (default 100); use at least 71 to run the current
integer from 1–100 on AWS (default 100); use at least 68 to run the current
complete catalog in one wave. The fallback runner retains its 1–57 limit and
default of 32. Multi-turn steps are sequential inside their cell while
independent cells overlap. Artifacts and merged HTML/JUnit/normalized reports

View File

@ -2,6 +2,7 @@ import { describe, expect, it } from "vitest";
import {
runnerEnvironments,
runnerMatrix,
openRouterBreadthExcludedModelIds,
openRouterBreadthProfiles,
openRouterBreadthTasks,
localIntegrityTasks,
@ -21,16 +22,16 @@ import {
describe("runner E2E catalog", () => {
it("validates the core, local-integrity, and breadth suites", () => {
expect(runnerProfiles).toHaveLength(7);
expect(openRouterBreadthProfiles).toHaveLength(5);
expect(openRouterBreadthProfiles).toHaveLength(4);
expect(runnerEnvironments).toHaveLength(2);
expect(runnerTasks).toHaveLength(3);
expect(localIntegrityTasks).toHaveLength(2);
expect(openRouterBreadthTasks).toHaveLength(3);
expect(runnerSuites.map((suite) => suite.expectedMatrixSize)).toEqual([
42, 14, 15,
42, 14, 12,
]);
expect(validateRunnerCatalog()).toHaveLength(71);
expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(71);
expect(validateRunnerCatalog()).toHaveLength(68);
expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(68);
expect(
runnerMatrix.filter((entry) => entry.suite.id === "core-compatibility"),
).toHaveLength(42);
@ -43,19 +44,20 @@ describe("runner E2E catalog", () => {
runnerMatrix.filter(
(entry) => entry.suite.id === "openrouter-model-breadth",
),
).toHaveLength(15);
).toHaveLength(12);
expect(
runnerMatrix.reduce(
(total, execution) => total + execution.task.expectedRunCount,
0,
),
).toBe(123);
).toBe(118);
});
it("derives five local native OpenCode profiles from the ranked snapshot", () => {
it("derives the qualified local native OpenCode profiles from the ranked snapshot", () => {
expect(openRouterBreadthExcludedModelIds).toEqual(["xiaomi/mimo-v2.5"]);
expect(
openRouterBreadthProfiles.map((profile) => profile.ranking?.rank),
).toEqual([1, 2, 3, 4, 5]);
).toEqual([1, 3, 4, 5]);
expect(
openRouterBreadthProfiles.every(
(profile) =>
@ -313,7 +315,7 @@ describe("runner E2E selectors", () => {
const selected = selectRunnerExecutions(
parseRunnerSelectors(["--suite", "openrouter-model-breadth"]),
);
expect(selected).toHaveLength(15);
expect(selected).toHaveLength(12);
expect(
selected.every(
(entry) =>
@ -350,9 +352,9 @@ describe("runner E2E selectors", () => {
const jobs = buildMatrixJobs(
selectRunnerExecutions(parseRunnerSelectors(["--all"])),
);
expect(jobs).toHaveLength(71);
expect(jobs).toHaveLength(68);
expect(jobs.filter((job) => job.needsDaytona)).toHaveLength(21);
expect(new Set(jobs.map((job) => job.executionId)).size).toBe(71);
expect(new Set(jobs.map((job) => job.executionId)).size).toBe(68);
expect(
jobs.every((job) =>
runnerMatrix.some(

View File

@ -262,28 +262,37 @@ export const runnerProfiles: readonly RunnerProfileFixture[] = [
}),
] as const;
export const openRouterBreadthExcludedModelIds = ["xiaomi/mimo-v2.5"] as const;
const openRouterBreadthExcludedModelIdSet = new Set<string>(
openRouterBreadthExcludedModelIds,
);
export const openRouterBreadthProfiles: readonly RunnerProfileFixture[] =
openRouterRankingSnapshot.models.map((rankedModel) =>
nativeProfile({
id: openRouterProfileId(rankedModel.id),
label: `#${rankedModel.rank} ${rankedModel.name}`,
provider: "opencode",
model: `openrouter/${rankedModel.id}`,
credential: "OPENROUTER_API_KEY",
supportedEnvironments: ["local"],
modelQualification: {
source: "openrouter_rankings_snapshot",
qualificationId: `${openRouterRankingSnapshot.snapshotId}:${rankedModel.rank}`,
},
ranking: {
rank: rankedModel.rank,
canonicalModelId: rankedModel.id,
snapshotId: openRouterRankingSnapshot.snapshotId,
capturedAt: openRouterRankingSnapshot.capturedAt,
sourceUrl: openRouterRankingSnapshot.sourceUrl,
},
}),
);
openRouterRankingSnapshot.models
.filter(
(rankedModel) => !openRouterBreadthExcludedModelIdSet.has(rankedModel.id),
)
.map((rankedModel) =>
nativeProfile({
id: openRouterProfileId(rankedModel.id),
label: `#${rankedModel.rank} ${rankedModel.name}`,
provider: "opencode",
model: `openrouter/${rankedModel.id}`,
credential: "OPENROUTER_API_KEY",
supportedEnvironments: ["local"],
modelQualification: {
source: "openrouter_rankings_snapshot",
qualificationId: `${openRouterRankingSnapshot.snapshotId}:${rankedModel.rank}`,
},
ranking: {
rank: rankedModel.rank,
canonicalModelId: rankedModel.id,
snapshotId: openRouterRankingSnapshot.snapshotId,
capturedAt: openRouterRankingSnapshot.capturedAt,
sourceUrl: openRouterRankingSnapshot.sourceUrl,
},
}),
);
function requiredDaytonaSecret(input: EnvironmentFixtureBuildInput) {
const apiKey = input.secretRefs.DAYTONA_API_KEY;
@ -734,12 +743,13 @@ export const runnerSuites: readonly RunnerSuiteFixture[] = [
profiles: openRouterBreadthProfiles,
environments: [localEnvironment],
tasks: openRouterBreadthTasks,
expectedMatrixSize: 15,
expectedMatrixSize: 12,
definitionMetadata: {
rankingSnapshotId: openRouterRankingSnapshot.snapshotId,
rankingContentHash: openRouterRankingSnapshot.contentHash,
rankingCapturedAt: openRouterRankingSnapshot.capturedAt,
rankingSourceUrl: openRouterRankingSnapshot.sourceUrl,
excludedModelIds: openRouterBreadthExcludedModelIds,
},
},
] as const;
@ -944,8 +954,8 @@ export function validateRunnerCatalog(): MatrixExecution[] {
);
}
}
if (matrix.length !== 71)
throw new Error(`Expected 71 runner executions; received ${matrix.length}`);
if (matrix.length !== 68)
throw new Error(`Expected 68 runner executions; received ${matrix.length}`);
return matrix;
}

View File

@ -68,15 +68,15 @@ describe("runner E2E campaign history", () => {
expected: breadth.map((execution) => execution.id),
results: breadth.map((execution) => result(execution, "passed")),
});
expect(campaign).toMatchObject({ complete: false, passed: 15, failed: 0 });
expect(campaign).toMatchObject({ complete: false, passed: 12, failed: 0 });
expect(campaign.suites[0]).toMatchObject({
suiteId: "openrouter-model-breadth",
complete: true,
selected: 15,
selected: 12,
});
expect(campaign.billing).toMatchObject({
reportedLlmCostUsd: 0.15,
llm: { inputTokens: 1_500, outputTokens: 375 },
reportedLlmCostUsd: 0.12,
llm: { inputTokens: 1_200, outputTokens: 300 },
});
const history = mergeRunnerHistory(
emptyRunnerHistory(),
@ -151,8 +151,8 @@ describe("runner E2E campaign history", () => {
expect(index).toContain("Runner E2E campaigns");
expect(index).toContain("complete-green");
expect(index).toContain("complete-red");
expect(index).toContain("71/71 passed");
expect(index).toContain("70/71 passed");
expect(index).toContain("68/68 passed");
expect(index).toContain("67/68 passed");
expect(index).toContain("Open report&nbsp;→");
expect(index).toContain(
"Visual evidence remains in access-controlled workflow artifacts",