test(runner): retire Xiaomi breadth cells
This commit is contained in:
parent
05c7f3a857
commit
e228846d4a
|
|
@ -80,9 +80,11 @@ workflows: 42 cells. Its cases are:
|
|||
same Plan, browser acceptance of the new revision, and verified execution;
|
||||
- `ask-question`: a direct answer from a task created in Ask mode.
|
||||
|
||||
`openrouter-model-breadth` (**OpenRouter Model Breadth**) is five models from
|
||||
the tracked weekly tool-capable ranking snapshot × native OpenCode × local ×
|
||||
three workflows: 15 cells. Its cases are:
|
||||
`openrouter-model-breadth` (**OpenRouter Model Breadth**) is four qualified
|
||||
models from the tracked weekly tool-capable ranking snapshot × native OpenCode
|
||||
× local × three workflows: 12 cells. Xiaomi MiMo V2.5 remains recorded in the
|
||||
immutable ranking snapshot but is excluded from paid qualification because its
|
||||
latency repeatedly exhausts the cell deadline. Its cases are:
|
||||
|
||||
- `hello-complete`: a basic nonce response and explicit Done transition;
|
||||
- `question-resume-complete`: one structured question, browser selection of
|
||||
|
|
@ -98,7 +100,7 @@ duplicating the final response. The second workflow restarts the isolated
|
|||
Paperclip server while the interaction is waiting, reloads that state, and
|
||||
then resumes it. The suite has no Daytona cells.
|
||||
|
||||
The complete catalog is 71 cells and 123 expected paid agent turns. Follow-up
|
||||
The complete catalog is 68 cells and 118 expected paid agent turns. Follow-up
|
||||
steps remain ordered within their cell; all other cells are independent.
|
||||
Narrow selectors are strongly recommended while developing fixtures.
|
||||
|
||||
|
|
@ -290,7 +292,7 @@ Set `RUNNER_E2E_AWS_ENABLED=true` to route paid cells to the repository-scoped
|
|||
ephemeral AWS RunsOn fleet selected by
|
||||
`runs-on/fleet=paperclip-public-pr-x64/env=public-ci`. Any other value uses the
|
||||
proven GitHub-hosted `ubuntu-latest` target. Set `RUNNER_E2E_MAX_PARALLEL` to an
|
||||
integer from 1–100 on AWS (default 100); use at least 71 to run the current
|
||||
integer from 1–100 on AWS (default 100); use at least 68 to run the current
|
||||
complete catalog in one wave. The fallback runner retains its 1–57 limit and
|
||||
default of 32. Multi-turn steps are sequential inside their cell while
|
||||
independent cells overlap. Artifacts and merged HTML/JUnit/normalized reports
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@ import { describe, expect, it } from "vitest";
|
|||
import {
|
||||
runnerEnvironments,
|
||||
runnerMatrix,
|
||||
openRouterBreadthExcludedModelIds,
|
||||
openRouterBreadthProfiles,
|
||||
openRouterBreadthTasks,
|
||||
localIntegrityTasks,
|
||||
|
|
@ -21,16 +22,16 @@ import {
|
|||
describe("runner E2E catalog", () => {
|
||||
it("validates the core, local-integrity, and breadth suites", () => {
|
||||
expect(runnerProfiles).toHaveLength(7);
|
||||
expect(openRouterBreadthProfiles).toHaveLength(5);
|
||||
expect(openRouterBreadthProfiles).toHaveLength(4);
|
||||
expect(runnerEnvironments).toHaveLength(2);
|
||||
expect(runnerTasks).toHaveLength(3);
|
||||
expect(localIntegrityTasks).toHaveLength(2);
|
||||
expect(openRouterBreadthTasks).toHaveLength(3);
|
||||
expect(runnerSuites.map((suite) => suite.expectedMatrixSize)).toEqual([
|
||||
42, 14, 15,
|
||||
42, 14, 12,
|
||||
]);
|
||||
expect(validateRunnerCatalog()).toHaveLength(71);
|
||||
expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(71);
|
||||
expect(validateRunnerCatalog()).toHaveLength(68);
|
||||
expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(68);
|
||||
expect(
|
||||
runnerMatrix.filter((entry) => entry.suite.id === "core-compatibility"),
|
||||
).toHaveLength(42);
|
||||
|
|
@ -43,19 +44,20 @@ describe("runner E2E catalog", () => {
|
|||
runnerMatrix.filter(
|
||||
(entry) => entry.suite.id === "openrouter-model-breadth",
|
||||
),
|
||||
).toHaveLength(15);
|
||||
).toHaveLength(12);
|
||||
expect(
|
||||
runnerMatrix.reduce(
|
||||
(total, execution) => total + execution.task.expectedRunCount,
|
||||
0,
|
||||
),
|
||||
).toBe(123);
|
||||
).toBe(118);
|
||||
});
|
||||
|
||||
it("derives five local native OpenCode profiles from the ranked snapshot", () => {
|
||||
it("derives the qualified local native OpenCode profiles from the ranked snapshot", () => {
|
||||
expect(openRouterBreadthExcludedModelIds).toEqual(["xiaomi/mimo-v2.5"]);
|
||||
expect(
|
||||
openRouterBreadthProfiles.map((profile) => profile.ranking?.rank),
|
||||
).toEqual([1, 2, 3, 4, 5]);
|
||||
).toEqual([1, 3, 4, 5]);
|
||||
expect(
|
||||
openRouterBreadthProfiles.every(
|
||||
(profile) =>
|
||||
|
|
@ -313,7 +315,7 @@ describe("runner E2E selectors", () => {
|
|||
const selected = selectRunnerExecutions(
|
||||
parseRunnerSelectors(["--suite", "openrouter-model-breadth"]),
|
||||
);
|
||||
expect(selected).toHaveLength(15);
|
||||
expect(selected).toHaveLength(12);
|
||||
expect(
|
||||
selected.every(
|
||||
(entry) =>
|
||||
|
|
@ -350,9 +352,9 @@ describe("runner E2E selectors", () => {
|
|||
const jobs = buildMatrixJobs(
|
||||
selectRunnerExecutions(parseRunnerSelectors(["--all"])),
|
||||
);
|
||||
expect(jobs).toHaveLength(71);
|
||||
expect(jobs).toHaveLength(68);
|
||||
expect(jobs.filter((job) => job.needsDaytona)).toHaveLength(21);
|
||||
expect(new Set(jobs.map((job) => job.executionId)).size).toBe(71);
|
||||
expect(new Set(jobs.map((job) => job.executionId)).size).toBe(68);
|
||||
expect(
|
||||
jobs.every((job) =>
|
||||
runnerMatrix.some(
|
||||
|
|
|
|||
|
|
@ -262,28 +262,37 @@ export const runnerProfiles: readonly RunnerProfileFixture[] = [
|
|||
}),
|
||||
] as const;
|
||||
|
||||
export const openRouterBreadthExcludedModelIds = ["xiaomi/mimo-v2.5"] as const;
|
||||
const openRouterBreadthExcludedModelIdSet = new Set<string>(
|
||||
openRouterBreadthExcludedModelIds,
|
||||
);
|
||||
|
||||
export const openRouterBreadthProfiles: readonly RunnerProfileFixture[] =
|
||||
openRouterRankingSnapshot.models.map((rankedModel) =>
|
||||
nativeProfile({
|
||||
id: openRouterProfileId(rankedModel.id),
|
||||
label: `#${rankedModel.rank} ${rankedModel.name}`,
|
||||
provider: "opencode",
|
||||
model: `openrouter/${rankedModel.id}`,
|
||||
credential: "OPENROUTER_API_KEY",
|
||||
supportedEnvironments: ["local"],
|
||||
modelQualification: {
|
||||
source: "openrouter_rankings_snapshot",
|
||||
qualificationId: `${openRouterRankingSnapshot.snapshotId}:${rankedModel.rank}`,
|
||||
},
|
||||
ranking: {
|
||||
rank: rankedModel.rank,
|
||||
canonicalModelId: rankedModel.id,
|
||||
snapshotId: openRouterRankingSnapshot.snapshotId,
|
||||
capturedAt: openRouterRankingSnapshot.capturedAt,
|
||||
sourceUrl: openRouterRankingSnapshot.sourceUrl,
|
||||
},
|
||||
}),
|
||||
);
|
||||
openRouterRankingSnapshot.models
|
||||
.filter(
|
||||
(rankedModel) => !openRouterBreadthExcludedModelIdSet.has(rankedModel.id),
|
||||
)
|
||||
.map((rankedModel) =>
|
||||
nativeProfile({
|
||||
id: openRouterProfileId(rankedModel.id),
|
||||
label: `#${rankedModel.rank} ${rankedModel.name}`,
|
||||
provider: "opencode",
|
||||
model: `openrouter/${rankedModel.id}`,
|
||||
credential: "OPENROUTER_API_KEY",
|
||||
supportedEnvironments: ["local"],
|
||||
modelQualification: {
|
||||
source: "openrouter_rankings_snapshot",
|
||||
qualificationId: `${openRouterRankingSnapshot.snapshotId}:${rankedModel.rank}`,
|
||||
},
|
||||
ranking: {
|
||||
rank: rankedModel.rank,
|
||||
canonicalModelId: rankedModel.id,
|
||||
snapshotId: openRouterRankingSnapshot.snapshotId,
|
||||
capturedAt: openRouterRankingSnapshot.capturedAt,
|
||||
sourceUrl: openRouterRankingSnapshot.sourceUrl,
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
||||
function requiredDaytonaSecret(input: EnvironmentFixtureBuildInput) {
|
||||
const apiKey = input.secretRefs.DAYTONA_API_KEY;
|
||||
|
|
@ -734,12 +743,13 @@ export const runnerSuites: readonly RunnerSuiteFixture[] = [
|
|||
profiles: openRouterBreadthProfiles,
|
||||
environments: [localEnvironment],
|
||||
tasks: openRouterBreadthTasks,
|
||||
expectedMatrixSize: 15,
|
||||
expectedMatrixSize: 12,
|
||||
definitionMetadata: {
|
||||
rankingSnapshotId: openRouterRankingSnapshot.snapshotId,
|
||||
rankingContentHash: openRouterRankingSnapshot.contentHash,
|
||||
rankingCapturedAt: openRouterRankingSnapshot.capturedAt,
|
||||
rankingSourceUrl: openRouterRankingSnapshot.sourceUrl,
|
||||
excludedModelIds: openRouterBreadthExcludedModelIds,
|
||||
},
|
||||
},
|
||||
] as const;
|
||||
|
|
@ -944,8 +954,8 @@ export function validateRunnerCatalog(): MatrixExecution[] {
|
|||
);
|
||||
}
|
||||
}
|
||||
if (matrix.length !== 71)
|
||||
throw new Error(`Expected 71 runner executions; received ${matrix.length}`);
|
||||
if (matrix.length !== 68)
|
||||
throw new Error(`Expected 68 runner executions; received ${matrix.length}`);
|
||||
return matrix;
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -68,15 +68,15 @@ describe("runner E2E campaign history", () => {
|
|||
expected: breadth.map((execution) => execution.id),
|
||||
results: breadth.map((execution) => result(execution, "passed")),
|
||||
});
|
||||
expect(campaign).toMatchObject({ complete: false, passed: 15, failed: 0 });
|
||||
expect(campaign).toMatchObject({ complete: false, passed: 12, failed: 0 });
|
||||
expect(campaign.suites[0]).toMatchObject({
|
||||
suiteId: "openrouter-model-breadth",
|
||||
complete: true,
|
||||
selected: 15,
|
||||
selected: 12,
|
||||
});
|
||||
expect(campaign.billing).toMatchObject({
|
||||
reportedLlmCostUsd: 0.15,
|
||||
llm: { inputTokens: 1_500, outputTokens: 375 },
|
||||
reportedLlmCostUsd: 0.12,
|
||||
llm: { inputTokens: 1_200, outputTokens: 300 },
|
||||
});
|
||||
const history = mergeRunnerHistory(
|
||||
emptyRunnerHistory(),
|
||||
|
|
@ -151,8 +151,8 @@ describe("runner E2E campaign history", () => {
|
|||
expect(index).toContain("Runner E2E campaigns");
|
||||
expect(index).toContain("complete-green");
|
||||
expect(index).toContain("complete-red");
|
||||
expect(index).toContain("71/71 passed");
|
||||
expect(index).toContain("70/71 passed");
|
||||
expect(index).toContain("68/68 passed");
|
||||
expect(index).toContain("67/68 passed");
|
||||
expect(index).toContain("Open report →");
|
||||
expect(index).toContain(
|
||||
"Visual evidence remains in access-controlled workflow artifacts",
|
||||
|
|
|
|||
Loading…
Reference in New Issue