test(runner): quarantine unqualified Tencent plan cell

This commit is contained in:
Dotta 2026-09-03 14:41:18 -05:00
parent 89e41f9768
commit f4686d9faa
4 changed files with 44 additions and 27 deletions

View File

@ -82,12 +82,16 @@ workflows: 42 cells. Its cases are:
`openrouter-model-breadth` (**OpenRouter Model Breadth**) is four qualified
models from the tracked weekly tool-capable ranking snapshot × native OpenCode
× local, with 11 supported model/workflow cells. Xiaomi MiMo V2.5 remains
× local, with 10 supported model/workflow cells. Xiaomi MiMo V2.5 remains
recorded in the immutable ranking snapshot but is excluded from paid
qualification because its latency repeatedly exhausts the cell deadline.
DeepSeek V4 Flash remains qualified for hello and question/resume, but its Plan
cell is excluded after three successful semantic completions consistently
ignored the required exact final response. Its cases are:
ignored the required exact final response. Tencent HY3 likewise remains
qualified for hello and question/resume, but its Plan cell is excluded after
two fresh attempts completed every durable Plan and finalization operation yet
consistently replaced the required exact visible terminal marker with prose.
Its cases are:
- `hello-complete`: a basic nonce response and explicit Done transition;
- `question-resume-complete`: one structured question, browser selection of
@ -103,9 +107,10 @@ duplicating the final response. The second workflow restarts the isolated
Paperclip server while the interaction is waiting, reloads that state, and
then resumes it. The suite has no Daytona cells.
The complete catalog is 67 cells and 116 expected paid agent turns. Follow-up
steps remain ordered within their cell; all other cells are independent.
Narrow selectors are strongly recommended while developing fixtures.
The complete catalog is 66 cells (45 local and 21 Daytona) and 114 expected
paid agent turns. Follow-up steps remain ordered within their cell; all other
cells are independent. Narrow selectors are strongly recommended while
developing fixtures.
`--suite`, `--group`, `--profile`, `--environment`, and `--case` are repeatable. Repeated
values in one dimension use OR semantics; dimensions and repeated groups use
@ -330,7 +335,7 @@ Set `RUNNER_E2E_AWS_ENABLED=true` to route paid cells to the repository-scoped
ephemeral AWS RunsOn fleet selected by
`runs-on/fleet=paperclip-public-pr-x64/env=public-ci`. Any other value uses the
proven GitHub-hosted `ubuntu-latest` target. Set `RUNNER_E2E_MAX_PARALLEL` to an
integer from 1–100 on AWS (default 100); use at least 67 to run the current
integer from 1–100 on AWS (default 100); use at least 66 to run the current
complete catalog in one wave. The fallback runner retains its 1–57 limit and
default of 32. Multi-turn steps are sequential inside their cell while
independent cells overlap. Artifacts and merged HTML/JUnit/normalized reports

View File

@ -29,10 +29,10 @@ describe("runner E2E catalog", () => {
expect(localIntegrityTasks).toHaveLength(2);
expect(openRouterBreadthTasks).toHaveLength(3);
expect(runnerSuites.map((suite) => suite.expectedMatrixSize)).toEqual([
42, 14, 11,
42, 14, 10,
]);
expect(validateRunnerCatalog()).toHaveLength(67);
expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(67);
expect(validateRunnerCatalog()).toHaveLength(66);
expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(66);
expect(
runnerMatrix.filter((entry) => entry.suite.id === "core-compatibility"),
).toHaveLength(42);
@ -45,26 +45,36 @@ describe("runner E2E catalog", () => {
runnerMatrix.filter(
(entry) => entry.suite.id === "openrouter-model-breadth",
),
).toHaveLength(11);
).toHaveLength(10);
expect(
runnerMatrix.reduce(
(total, execution) => total + execution.task.expectedRunCount,
0,
),
).toBe(116);
).toBe(114);
});
it("derives the qualified local native OpenCode profiles from the ranked snapshot", () => {
expect(openRouterBreadthExcludedModelIds).toEqual(["xiaomi/mimo-v2.5"]);
expect(openRouterBreadthExcludedExecutionIds).toEqual([
"openrouter-model-breadth.openrouter-deepseek-deepseek-v4-flash-0731.local.plan-approve-complete",
"openrouter-model-breadth.openrouter-tencent-hy3.local.plan-approve-complete",
]);
expect(
runnerMatrix.some(
(execution) =>
execution.id === openRouterBreadthExcludedExecutionIds[0],
openRouterBreadthExcludedExecutionIds.every(
(excludedExecutionId) =>
!runnerMatrix.some(
(execution) => execution.id === excludedExecutionId,
),
),
).toBe(false);
).toBe(true);
expect(
runnerMatrix
.filter(
(execution) => execution.profile.id === "openrouter-tencent-hy3",
)
.map((execution) => execution.task.id),
).toEqual(["hello-complete", "question-resume-complete"]);
expect(
openRouterBreadthProfiles.map((profile) => profile.ranking?.rank),
).toEqual([1, 3, 4, 5]);
@ -323,7 +333,7 @@ describe("runner E2E selectors", () => {
const selected = selectRunnerExecutions(
parseRunnerSelectors(["--suite", "openrouter-model-breadth"]),
);
expect(selected).toHaveLength(11);
expect(selected).toHaveLength(10);
expect(
selected.every(
(entry) =>
@ -360,9 +370,10 @@ describe("runner E2E selectors", () => {
const jobs = buildMatrixJobs(
selectRunnerExecutions(parseRunnerSelectors(["--all"])),
);
expect(jobs).toHaveLength(67);
expect(jobs).toHaveLength(66);
expect(jobs.filter((job) => job.needsDaytona)).toHaveLength(21);
expect(new Set(jobs.map((job) => job.executionId)).size).toBe(67);
expect(jobs.filter((job) => !job.needsDaytona)).toHaveLength(45);
expect(new Set(jobs.map((job) => job.executionId)).size).toBe(66);
expect(
jobs.every((job) =>
runnerMatrix.some(

View File

@ -265,6 +265,7 @@ export const runnerProfiles: readonly RunnerProfileFixture[] = [
export const openRouterBreadthExcludedModelIds = ["xiaomi/mimo-v2.5"] as const;
export const openRouterBreadthExcludedExecutionIds = [
"openrouter-model-breadth.openrouter-deepseek-deepseek-v4-flash-0731.local.plan-approve-complete",
"openrouter-model-breadth.openrouter-tencent-hy3.local.plan-approve-complete",
] as const;
const openRouterBreadthExcludedModelIdSet = new Set<string>(
openRouterBreadthExcludedModelIds,
@ -747,7 +748,7 @@ export const runnerSuites: readonly RunnerSuiteFixture[] = [
environments: [localEnvironment],
tasks: openRouterBreadthTasks,
excludedExecutionIds: openRouterBreadthExcludedExecutionIds,
expectedMatrixSize: 11,
expectedMatrixSize: 10,
definitionMetadata: {
rankingSnapshotId: openRouterRankingSnapshot.snapshotId,
rankingContentHash: openRouterRankingSnapshot.contentHash,
@ -963,8 +964,8 @@ export function validateRunnerCatalog(): MatrixExecution[] {
);
}
}
if (matrix.length !== 67)
throw new Error(`Expected 67 runner executions; received ${matrix.length}`);
if (matrix.length !== 66)
throw new Error(`Expected 66 runner executions; received ${matrix.length}`);
return matrix;
}

View File

@ -68,16 +68,16 @@ describe("runner E2E campaign history", () => {
expected: breadth.map((execution) => execution.id),
results: breadth.map((execution) => result(execution, "passed")),
});
expect(campaign).toMatchObject({ complete: false, passed: 11, failed: 0 });
expect(campaign).toMatchObject({ complete: false, passed: 10, failed: 0 });
expect(campaign.suites[0]).toMatchObject({
suiteId: "openrouter-model-breadth",
complete: true,
selected: 11,
selected: 10,
});
expect(campaign.billing).toMatchObject({
llm: { inputTokens: 1_100, outputTokens: 275 },
llm: { inputTokens: 1_000, outputTokens: 250 },
});
expect(campaign.billing.reportedLlmCostUsd).toBeCloseTo(0.11, 10);
expect(campaign.billing.reportedLlmCostUsd).toBeCloseTo(0.1, 10);
const history = mergeRunnerHistory(
emptyRunnerHistory(),
campaignHistoryRecord(campaign, "https://history.example/runner-e2e"),
@ -151,8 +151,8 @@ describe("runner E2E campaign history", () => {
expect(index).toContain("Runner E2E campaigns");
expect(index).toContain("complete-green");
expect(index).toContain("complete-red");
expect(index).toContain("67/67 passed");
expect(index).toContain("66/67 passed");
expect(index).toContain("66/66 passed");
expect(index).toContain("65/66 passed");
expect(index).toContain("Open report&nbsp;→");
expect(index).toContain(
"Visual evidence remains in access-controlled workflow artifacts",