test(runner): quarantine unqualified Tencent plan cell
This commit is contained in:
parent
89e41f9768
commit
f4686d9faa
|
|
@ -82,12 +82,16 @@ workflows: 42 cells. Its cases are:
|
|||
|
||||
`openrouter-model-breadth` (**OpenRouter Model Breadth**) is four qualified
|
||||
models from the tracked weekly tool-capable ranking snapshot × native OpenCode
|
||||
× local, with 11 supported model/workflow cells. Xiaomi MiMo V2.5 remains
|
||||
× local, with 10 supported model/workflow cells. Xiaomi MiMo V2.5 remains
|
||||
recorded in the immutable ranking snapshot but is excluded from paid
|
||||
qualification because its latency repeatedly exhausts the cell deadline.
|
||||
DeepSeek V4 Flash remains qualified for hello and question/resume, but its Plan
|
||||
cell is excluded after three successful semantic completions consistently
|
||||
ignored the required exact final response. Its cases are:
|
||||
ignored the required exact final response. Tencent HY3 likewise remains
|
||||
qualified for hello and question/resume, but its Plan cell is excluded after
|
||||
two fresh attempts completed every durable Plan and finalization operation yet
|
||||
consistently replaced the required exact visible terminal marker with prose.
|
||||
Its cases are:
|
||||
|
||||
- `hello-complete`: a basic nonce response and explicit Done transition;
|
||||
- `question-resume-complete`: one structured question, browser selection of
|
||||
|
|
@ -103,9 +107,10 @@ duplicating the final response. The second workflow restarts the isolated
|
|||
Paperclip server while the interaction is waiting, reloads that state, and
|
||||
then resumes it. The suite has no Daytona cells.
|
||||
|
||||
The complete catalog is 67 cells and 116 expected paid agent turns. Follow-up
|
||||
steps remain ordered within their cell; all other cells are independent.
|
||||
Narrow selectors are strongly recommended while developing fixtures.
|
||||
The complete catalog is 66 cells (45 local and 21 Daytona) and 114 expected
|
||||
paid agent turns. Follow-up steps remain ordered within their cell; all other
|
||||
cells are independent. Narrow selectors are strongly recommended while
|
||||
developing fixtures.
|
||||
|
||||
`--suite`, `--group`, `--profile`, `--environment`, and `--case` are repeatable. Repeated
|
||||
values in one dimension use OR semantics; dimensions and repeated groups use
|
||||
|
|
@ -330,7 +335,7 @@ Set `RUNNER_E2E_AWS_ENABLED=true` to route paid cells to the repository-scoped
|
|||
ephemeral AWS RunsOn fleet selected by
|
||||
`runs-on/fleet=paperclip-public-pr-x64/env=public-ci`. Any other value uses the
|
||||
proven GitHub-hosted `ubuntu-latest` target. Set `RUNNER_E2E_MAX_PARALLEL` to an
|
||||
integer from 1–100 on AWS (default 100); use at least 67 to run the current
|
||||
integer from 1–100 on AWS (default 100); use at least 66 to run the current
|
||||
complete catalog in one wave. The fallback runner retains its 1–57 limit and
|
||||
default of 32. Multi-turn steps are sequential inside their cell while
|
||||
independent cells overlap. Artifacts and merged HTML/JUnit/normalized reports
|
||||
|
|
|
|||
|
|
@ -29,10 +29,10 @@ describe("runner E2E catalog", () => {
|
|||
expect(localIntegrityTasks).toHaveLength(2);
|
||||
expect(openRouterBreadthTasks).toHaveLength(3);
|
||||
expect(runnerSuites.map((suite) => suite.expectedMatrixSize)).toEqual([
|
||||
42, 14, 11,
|
||||
42, 14, 10,
|
||||
]);
|
||||
expect(validateRunnerCatalog()).toHaveLength(67);
|
||||
expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(67);
|
||||
expect(validateRunnerCatalog()).toHaveLength(66);
|
||||
expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(66);
|
||||
expect(
|
||||
runnerMatrix.filter((entry) => entry.suite.id === "core-compatibility"),
|
||||
).toHaveLength(42);
|
||||
|
|
@ -45,26 +45,36 @@ describe("runner E2E catalog", () => {
|
|||
runnerMatrix.filter(
|
||||
(entry) => entry.suite.id === "openrouter-model-breadth",
|
||||
),
|
||||
).toHaveLength(11);
|
||||
).toHaveLength(10);
|
||||
expect(
|
||||
runnerMatrix.reduce(
|
||||
(total, execution) => total + execution.task.expectedRunCount,
|
||||
0,
|
||||
),
|
||||
).toBe(116);
|
||||
).toBe(114);
|
||||
});
|
||||
|
||||
it("derives the qualified local native OpenCode profiles from the ranked snapshot", () => {
|
||||
expect(openRouterBreadthExcludedModelIds).toEqual(["xiaomi/mimo-v2.5"]);
|
||||
expect(openRouterBreadthExcludedExecutionIds).toEqual([
|
||||
"openrouter-model-breadth.openrouter-deepseek-deepseek-v4-flash-0731.local.plan-approve-complete",
|
||||
"openrouter-model-breadth.openrouter-tencent-hy3.local.plan-approve-complete",
|
||||
]);
|
||||
expect(
|
||||
runnerMatrix.some(
|
||||
(execution) =>
|
||||
execution.id === openRouterBreadthExcludedExecutionIds[0],
|
||||
openRouterBreadthExcludedExecutionIds.every(
|
||||
(excludedExecutionId) =>
|
||||
!runnerMatrix.some(
|
||||
(execution) => execution.id === excludedExecutionId,
|
||||
),
|
||||
),
|
||||
).toBe(false);
|
||||
).toBe(true);
|
||||
expect(
|
||||
runnerMatrix
|
||||
.filter(
|
||||
(execution) => execution.profile.id === "openrouter-tencent-hy3",
|
||||
)
|
||||
.map((execution) => execution.task.id),
|
||||
).toEqual(["hello-complete", "question-resume-complete"]);
|
||||
expect(
|
||||
openRouterBreadthProfiles.map((profile) => profile.ranking?.rank),
|
||||
).toEqual([1, 3, 4, 5]);
|
||||
|
|
@ -323,7 +333,7 @@ describe("runner E2E selectors", () => {
|
|||
const selected = selectRunnerExecutions(
|
||||
parseRunnerSelectors(["--suite", "openrouter-model-breadth"]),
|
||||
);
|
||||
expect(selected).toHaveLength(11);
|
||||
expect(selected).toHaveLength(10);
|
||||
expect(
|
||||
selected.every(
|
||||
(entry) =>
|
||||
|
|
@ -360,9 +370,10 @@ describe("runner E2E selectors", () => {
|
|||
const jobs = buildMatrixJobs(
|
||||
selectRunnerExecutions(parseRunnerSelectors(["--all"])),
|
||||
);
|
||||
expect(jobs).toHaveLength(67);
|
||||
expect(jobs).toHaveLength(66);
|
||||
expect(jobs.filter((job) => job.needsDaytona)).toHaveLength(21);
|
||||
expect(new Set(jobs.map((job) => job.executionId)).size).toBe(67);
|
||||
expect(jobs.filter((job) => !job.needsDaytona)).toHaveLength(45);
|
||||
expect(new Set(jobs.map((job) => job.executionId)).size).toBe(66);
|
||||
expect(
|
||||
jobs.every((job) =>
|
||||
runnerMatrix.some(
|
||||
|
|
|
|||
|
|
@ -265,6 +265,7 @@ export const runnerProfiles: readonly RunnerProfileFixture[] = [
|
|||
export const openRouterBreadthExcludedModelIds = ["xiaomi/mimo-v2.5"] as const;
|
||||
export const openRouterBreadthExcludedExecutionIds = [
|
||||
"openrouter-model-breadth.openrouter-deepseek-deepseek-v4-flash-0731.local.plan-approve-complete",
|
||||
"openrouter-model-breadth.openrouter-tencent-hy3.local.plan-approve-complete",
|
||||
] as const;
|
||||
const openRouterBreadthExcludedModelIdSet = new Set<string>(
|
||||
openRouterBreadthExcludedModelIds,
|
||||
|
|
@ -747,7 +748,7 @@ export const runnerSuites: readonly RunnerSuiteFixture[] = [
|
|||
environments: [localEnvironment],
|
||||
tasks: openRouterBreadthTasks,
|
||||
excludedExecutionIds: openRouterBreadthExcludedExecutionIds,
|
||||
expectedMatrixSize: 11,
|
||||
expectedMatrixSize: 10,
|
||||
definitionMetadata: {
|
||||
rankingSnapshotId: openRouterRankingSnapshot.snapshotId,
|
||||
rankingContentHash: openRouterRankingSnapshot.contentHash,
|
||||
|
|
@ -963,8 +964,8 @@ export function validateRunnerCatalog(): MatrixExecution[] {
|
|||
);
|
||||
}
|
||||
}
|
||||
if (matrix.length !== 67)
|
||||
throw new Error(`Expected 67 runner executions; received ${matrix.length}`);
|
||||
if (matrix.length !== 66)
|
||||
throw new Error(`Expected 66 runner executions; received ${matrix.length}`);
|
||||
return matrix;
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -68,16 +68,16 @@ describe("runner E2E campaign history", () => {
|
|||
expected: breadth.map((execution) => execution.id),
|
||||
results: breadth.map((execution) => result(execution, "passed")),
|
||||
});
|
||||
expect(campaign).toMatchObject({ complete: false, passed: 11, failed: 0 });
|
||||
expect(campaign).toMatchObject({ complete: false, passed: 10, failed: 0 });
|
||||
expect(campaign.suites[0]).toMatchObject({
|
||||
suiteId: "openrouter-model-breadth",
|
||||
complete: true,
|
||||
selected: 11,
|
||||
selected: 10,
|
||||
});
|
||||
expect(campaign.billing).toMatchObject({
|
||||
llm: { inputTokens: 1_100, outputTokens: 275 },
|
||||
llm: { inputTokens: 1_000, outputTokens: 250 },
|
||||
});
|
||||
expect(campaign.billing.reportedLlmCostUsd).toBeCloseTo(0.11, 10);
|
||||
expect(campaign.billing.reportedLlmCostUsd).toBeCloseTo(0.1, 10);
|
||||
const history = mergeRunnerHistory(
|
||||
emptyRunnerHistory(),
|
||||
campaignHistoryRecord(campaign, "https://history.example/runner-e2e"),
|
||||
|
|
@ -151,8 +151,8 @@ describe("runner E2E campaign history", () => {
|
|||
expect(index).toContain("Runner E2E campaigns");
|
||||
expect(index).toContain("complete-green");
|
||||
expect(index).toContain("complete-red");
|
||||
expect(index).toContain("67/67 passed");
|
||||
expect(index).toContain("66/67 passed");
|
||||
expect(index).toContain("66/66 passed");
|
||||
expect(index).toContain("65/66 passed");
|
||||
expect(index).toContain("Open report →");
|
||||
expect(index).toContain(
|
||||
"Visual evidence remains in access-controlled workflow artifacts",
|
||||
|
|
|
|||
Loading…
Reference in New Issue