feat(evals): track comparable pass rates, costs, and source revisions over time

Co-Authored-By: Paperclip <noreply@paperclip.ing>
This commit is contained in:
Dotta 2026-09-06 21:14:09 -05:00
parent 40bad53aae
commit cad99dbf04
11 changed files with 420 additions and 69 deletions

View File

@ -56,9 +56,32 @@
.evalbook-site .status { display: inline-block; padding: var(--pit-space-1) var(--pit-space-2); border-radius: var(--pit-radius-sm); font-weight: 600; text-transform: uppercase; }
.evalbook-site small { display: block; margin-top: var(--pit-space-1); }
.evalbook-site footer { margin-top: var(--pit-space-5); font-size: var(--pit-text-xs); }
.evalbook-site .trend-group { padding: var(--pit-space-4); border: 1px solid var(--pit-border); border-radius: var(--pit-radius-sm); background: var(--pit-surface-sunken); }
.evalbook-site .trend-group > summary { color: var(--pit-foreground); font-weight: 600; }
.evalbook-site .trend-grid { display: grid; grid-template-columns: repeat(2, minmax(0, 1fr)); gap: var(--pit-space-5); }
.evalbook-site figure { min-width: 0; margin: var(--pit-space-4) 0; }
.evalbook-site figcaption { color: var(--pit-muted-foreground); font-size: var(--pit-text-xs); }
.evalbook-site .trend-chart { width: 100%; overflow: visible; }
.evalbook-site .trend-chart text { fill: var(--pit-muted-foreground); font: var(--pit-text-xs) var(--pit-font-mono); }
.evalbook-site .trend-axis { fill: none; stroke: var(--pit-border); }
.evalbook-site .trend-line { fill: none; stroke-width: 2; }
.evalbook-site :is(.trend-line, .trend-point).pass { stroke: var(--pit-success); }
.evalbook-site :is(.trend-line, .trend-point).all_attempts { stroke: var(--pit-live); }
.evalbook-site :is(.trend-line, .trend-point).final_attempts { stroke: var(--pit-accent); stroke-dasharray: 4 3; }
.evalbook-site .trend-point { fill: var(--pit-surface); stroke-width: 2; }
.evalbook-site .trend-point:hover { fill: var(--pit-foreground); }
.evalbook-site .pass-rate { display: block; margin-top: var(--pit-space-2); white-space: nowrap; }
.evalbook-site .history-table table { min-width: 68rem; }
.evalbook-site .history-table td:first-child { max-width: 22rem; }
.evalbook-site .history-table details { margin: var(--pit-space-2) 0; }
.evalbook-site .history-table li { margin: var(--pit-space-2) 0; }
.evalbook-site .incomplete { background: var(--pit-warning-surface); color: var(--pit-warning); }
@media (max-width: 760px) {
.evalbook-site main { padding: var(--pit-space-4); }
.evalbook-site .grid { grid-template-columns: 1fr; }
.evalbook-site .trend-grid { grid-template-columns: 1fr; }
.evalbook-site .trend-grid figure { overflow-x: auto; }
.evalbook-site .trend-chart { min-width: 30rem; }
.evalbook-site .matrix th:first-child { width: 12rem; min-width: 10rem; }
.evalbook-site .badge { white-space: normal; text-align: center; }
}

View File

@ -74,8 +74,10 @@ revisions do not change. This is not a new model qualification run. Future live
runs create chat reports automatically.
Refreshes are ordered by their render time in the history list, but keep the
original measurement timestamp and never replace the latest or latest-green
qualification pointers. Those two real-run records remain retained even when
many report revisions fill the history window.
qualification identities. The HTML history links each measurement to its newest
presentation, while retaining every original bundle and listing refreshes separately.
Refreshes also recover retry-inclusive estimated and provider-list costs from the
retained attempt records. They never add another model-cost measurement.
This is the provider-backed, one-turn protocol qualification layer in
`paperclipai/paperclip-evals/evals/paperclip-runner`. It is intentionally
@ -255,9 +257,27 @@ runner-protocol-evals/
Campaign files use immutable cache headers and a digest manifest. Reusing a
campaign ID with different bytes fails closed. Only the root history and
pointer files are mutable, and the publisher never deletes objects. The root
history retains at most 200 records, reserving one record for the latest green
campaign when it would otherwise fall outside that window so its pointer stays
valid.
history retains **all** run records; it no longer drops entries after 200 campaigns.
The history index includes:
- Pass-rate and cost timelines, grouped by identical cell/model/driver membership
and eval-suite SHA. Different suites and subsets cannot silently share a baseline.
- Regression and recovery lists against the previous matching run, linking to the
affected tests. Infrastructure failures stay distinct from behavior failures.
- Estimated cost and provider-reported list cost, shown separately, never added.
New campaigns include all retained attempts (including retries). Backfilled old
campaigns with only final-cell usage are labeled **historical final attempts only**.
Missing usage is unknown, not zero; partial totals use `≥` and display coverage.
- Exact Paperclip and eval-suite commit links (full SHA on hover), the source ref,
and the GitHub Actions run. These identify the code **evaluated**, not merely
the commit used to render an old report.
- A separate report-refresh list, excluded from trend points and model-spend totals.
`history.json` stores a versioned, derived `analytics` projection separately from
immutable campaign records. The publisher backfills missing analytics from each
original `campaign.json`; a refresh may enrich costs using retained raw attempts
only when its source metadata and scores still match the original record.
The publishing job uses dedicated `RUNNER_PROTOCOL_EVAL_HISTORY_*` variables
when present and falls back to the existing Runner E2E history role, region,

View File

@ -115,7 +115,7 @@
"test:capability-evals": "vitest run src/conformance/capability-eval-suite.test.ts",
"test:eval-slice": "pnpm run ensure:eval-build-deps && vitest run src/eval",
"test:runner-workflow-evals": "pnpm run ensure:eval-build-deps && node --test scripts/render-runner-workflow-evalbook.test.mjs && vitest run src/eval/workflow-evals.test.ts src/eval/live-workflow-executor.test.ts",
"test:runner-protocol-eval-publish": "node --test scripts/runner-protocol-eval-campaign.test.mjs scripts/publish-runner-protocol-eval-history.test.mjs scripts/runner-protocol-eval-workflow-security.test.mjs scripts/public-eval-chat.test.mjs",
"test:runner-protocol-eval-publish": "node --test scripts/runner-protocol-eval-campaign.test.mjs scripts/publish-runner-protocol-eval-history.test.mjs scripts/runner-protocol-eval-metrics.test.mjs scripts/runner-protocol-eval-workflow-security.test.mjs scripts/public-eval-chat.test.mjs",
"check:runner-workflow-traceability": "pnpm run build:typescript && node scripts/check-runner-workflow-traceability.mjs",
"report:capability-evals": "pnpm run build:typescript && node scripts/run-capability-eval-suite.mjs",
"report:capability-live-evals": "pnpm run build:typescript && cargo build --manifest-path runner/Cargo.toml --locked -p paperclip-runner-core --bin paperclip-runnerd && node scripts/run-capability-live-eval-matrix.mjs",

View File

@ -14,6 +14,9 @@ import {
import { tmpdir } from "node:os";
import { extname, join, relative, resolve, sep } from "node:path";
import { promisify } from "node:util";
import { enrichProtocolEvalHistory } from "./runner-protocol-eval-metrics.mjs";
import { renderProtocolEvalHistoryIndex } from "./runner-protocol-eval-history-view.mjs";
export { renderProtocolEvalHistoryIndex } from "./runner-protocol-eval-history-view.mjs";
import {
trustedViewerFiles,
validatePublicViewerPage,
@ -46,21 +49,11 @@ const ACTIVE_HTML_PATTERNS = [
/javascript\s*:/iu,
/(?:src|href)\s*=\s*["'](?:https?:)?\/\//iu,
];
const MAX_HISTORY_CAMPAIGNS = 200;
function json(value) {
return `${JSON.stringify(value, null, 2)}\n`;
}
function html(value) {
return String(value ?? "")
.replaceAll("&", "&amp;")
.replaceAll("<", "&lt;")
.replaceAll(">", "&gt;")
.replaceAll('"', "&quot;")
.replaceAll("'", "&#39;");
}
async function loadObject(path) {
const value = JSON.parse(await readFile(path, "utf8"));
if (value === null || Array.isArray(value) || typeof value !== "object") {
@ -338,18 +331,13 @@ export function mergeProtocolEvalHistory(history, record) {
qualifications.find(
(campaign) => campaign.complete && campaign.allPassed,
) ?? null;
const pointers = [...new Set([latest, latestGreen].filter(Boolean))];
const retained = campaigns
.filter((campaign) => !pointers.includes(campaign))
.slice(0, MAX_HISTORY_CAMPAIGNS - pointers.length)
.concat(pointers)
.sort(activityOrder);
return {
schema: history.schema,
updatedAt: new Date().toISOString(),
latestCampaignId: latest?.campaignId ?? null,
latestGreenCampaignId: latestGreen?.campaignId ?? null,
campaigns: retained,
campaigns,
...(history.analytics ? { analytics: history.analytics } : {}),
};
}
@ -383,46 +371,6 @@ export function buildProtocolEvalPointers(history) {
};
}
function date(value) {
return new Intl.DateTimeFormat("en-US", {
dateStyle: "medium",
timeStyle: "short",
timeZone: "UTC",
}).format(new Date(value));
}
export function renderProtocolEvalHistoryIndex(history, stylesheetHref) {
if (!/^campaigns\/gha-[a-z0-9-]+\/viewer\/assets\/[A-Za-z0-9._-]+\.css$/.test(stylesheetHref ?? ""))
throw new Error("History requires an immutable campaign's Runner Lab stylesheet");
const rows = history.campaigns.length
? history.campaigns
.map((campaign) => {
const status =
campaign.complete && campaign.allPassed ? "passed" : "failed";
const rosters = campaign.rosters
.map(
(roster) =>
`${html(roster.model)} · ${roster.passed}/${roster.selected}`,
)
.join("<br>");
return `<tr><td><a href="${html(campaign.publicUrl)}"><code>${html(campaign.campaignId)}</code></a><small>${html(date(campaign.generatedAt))} UTC</small>${campaign.reportRevision ? `<small>Report refresh · no new model calls · source ${html(campaign.reportRevision.sourceCampaignId)}</small>` : ""}</td><td><span class="status ${status}">${status}</span></td><td><strong>${html(campaign.totals.passed)}/${html(campaign.totals.selected)}</strong><small>${html(campaign.totals.behaviorFailures)} behavior · ${html(campaign.totals.infrastructureFailures)} infrastructure</small></td><td>${rosters}</td><td><code>${html(campaign.source?.paperclip?.sha?.slice(0, 8) ?? "unknown")}</code><small>evals ${html(campaign.source?.evals?.sha?.slice(0, 8) ?? "unknown")}</small></td><td><a href="${html(campaign.publicUrl)}">Open Evalbook →</a></td></tr>`;
})
.join("")
: '<tr><td colspan="6" class="empty">No campaigns have been published yet.</td></tr>';
const latest = history.campaigns.find(
(campaign) => campaign.campaignId === history.latestCampaignId,
);
const latestGreen = history.campaigns.find(
(campaign) => campaign.campaignId === history.latestGreenCampaignId,
);
return `<!doctype html>
<html lang="en"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width"><title>Runner protocol eval campaigns · Paperclip</title>
<link rel="stylesheet" href="${html(stylesheetHref)}"></head>
<body class="evalbook-site"><main><header class="top"><a href="index.html">paperclip-runner evals</a><span class="badge">Run history</span></header><h1>Runner protocol eval campaigns</h1><p class="muted">Versioned live-runner reports. Full provider evidence remains in access-controlled workflow artifacts.</p>
<nav class="pointers">${latest ? `<a href="${html(latest.publicUrl)}">Latest · ${html(latest.campaignId)}</a>` : ""}${latestGreen ? `<a href="${html(latestGreen.publicUrl)}">Latest green · ${html(latestGreen.campaignId)}</a>` : ""}</nav>
<div class="table"><table><thead><tr><th>Campaign</th><th>Status</th><th>Cells</th><th>Models / rosters</th><th>Source</th><th></th></tr></thead><tbody>${rows}</tbody></table></div><footer>Updated ${html(date(history.updatedAt))} UTC · Immutable campaign bundles · Canonical Evalbook layout with public-safe evidence projections</footer></main></body></html>`;
}
function awsObject(bucket, key) {
return `s3://${bucket}/${key}`;
}
@ -521,7 +469,7 @@ export async function publishProtocolEvalHistory({
join(tmpdir(), "runner-protocol-eval-history-"),
);
const historyKey = `${validatedDestination.prefix}/history.json`;
const history = mergeProtocolEvalHistory(
const mergedHistory = mergeProtocolEvalHistory(
(await downloadJson(
validatedDestination.bucket,
historyKey,
@ -532,6 +480,15 @@ export async function publishProtocolEvalHistory({
`${validatedDestination.publicBaseUrl}/${validatedDestination.prefix}`,
),
);
const history = await enrichProtocolEvalHistory(mergedHistory, {
currentCampaign: campaign,
loadCampaign: async (id) => {
if (!SAFE_CAMPAIGN.test(id)) throw new Error("Unsafe historical campaign ID");
return downloadJson(validatedDestination.bucket,
`${validatedDestination.prefix}/campaigns/${id}/campaign.json`,
join(temporary, `${id}.json`));
},
});
const campaignPrefix = `${validatedDestination.prefix}/campaigns/${campaign.campaignId}`;
const manifestKey = `${campaignPrefix}/bundle-manifest.json`;
const existing = await downloadJson(

View File

@ -211,7 +211,7 @@ test("retains immutable history and independent latest-green pointers", () => {
);
});
test("report refreshes never replace qualification pointers, including after retention", () => {
test("report refreshes never replace qualification pointers or evict older measurements", () => {
const record = (value) =>
protocolEvalHistoryRecord(
value,
@ -258,14 +258,14 @@ test("report refreshes never replace qualification pointers, including after ret
buildProtocolEvalPointers(history).latestGreen.campaign.campaignId,
"gha-42-1",
);
assert.equal(history.campaigns.length, 200);
assert.equal(history.campaigns.length, 207);
assert.match(
renderProtocolEvalHistoryIndex(history, "campaigns/gha-42-1-report-chat-v1/viewer/assets/index.css"),
/Report refresh · no new model calls/,
);
});
test("retains the latest green pointer outside the 200 newest campaigns", () => {
test("retains every run and the latest green pointer beyond 200 campaigns", () => {
const green = protocolEvalHistoryRecord(
campaign(),
"https://reports.example/runner-protocol-evals",
@ -287,7 +287,7 @@ test("retains the latest green pointer outside the 200 newest campaigns", () =>
);
}
assert.equal(history.campaigns.length, 200);
assert.equal(history.campaigns.length, 202);
assert.equal(history.latestCampaignId, "gha-243-1");
assert.equal(history.latestGreenCampaignId, "gha-42-1");
assert.equal(history.campaigns.at(-1).campaignId, "gha-42-1");

View File

@ -2,10 +2,11 @@
// Re-render immutable recorded evidence; this command never invokes a model.
import { execFileSync } from "node:child_process";
import { createHash } from "node:crypto";
import { mkdir, readFile, writeFile, lstat } from "node:fs/promises";
import { mkdir, readFile, writeFile, lstat, readdir } from "node:fs/promises";
import { join, resolve } from "node:path";
import { sanitizeProtocolEvalRuns } from "./runner-protocol-eval-campaign.mjs";
import { validatePublicProtocolEvalReport } from "./publish-runner-protocol-eval-history.mjs";
import { sumAttemptCosts } from "./runner-protocol-eval-metrics.mjs";
export async function refreshProtocolEvalReport({
sourceRoot,
@ -13,6 +14,7 @@ export async function refreshProtocolEvalReport({
viewerRoot,
outputRoot,
revision,
selection,
renderedAt = new Date().toISOString(),
}) {
if (!/^[a-z0-9][a-z0-9-]{0,39}$/.test(revision ?? ""))
@ -59,6 +61,10 @@ export async function refreshProtocolEvalReport({
);
const refreshed = {
...campaign,
...(selection ? { selection } : {}),
// Recover retry-inclusive cost from retained evidence, not just winning cells.
costs: sumAttemptCosts(await Promise.all((await readdir(runsRoot)).map(async (id) =>
JSON.parse(await readFile(join(runsRoot, id, "artifact.json"), "utf8")).usage))),
campaignId: `${campaign.campaignId}-report-${revision}`,
// A presentation refresh is not a new model measurement.
generatedAt: campaign.generatedAt,

View File

@ -11,6 +11,7 @@ import {
writeFile,
} from "node:fs/promises";
import { basename, dirname, join, relative, resolve, sep } from "node:path";
import { sumAttemptCosts } from "./runner-protocol-eval-metrics.mjs";
import {
publicChatView,
PUBLIC_CHAT_SCHEMA,
@ -207,6 +208,7 @@ export async function buildProtocolEvalCatalog({
schema: "paperclip.runner-protocol-eval.catalog/v1",
campaignId,
source,
selection: { kind: requested === null ? "maintained_full" : "subset", rosters: rosterSelection },
rosters,
cells,
matrices: shards.map((include) => ({ include })),
@ -438,6 +440,7 @@ export async function aggregateProtocolEvalCampaign({
}
const results = [];
const attemptUsages = [];
for (const cell of catalog.cells) {
const retained = retainedByCell.get(cell.cellId);
const attemptIds = retained?.attemptIds?.length
@ -451,6 +454,10 @@ export async function aggregateProtocolEvalCampaign({
}),
];
const finalAttemptId = attemptIds.at(-1);
for (const attemptId of attemptIds) {
const attempt = await loadObject(join(runsOut, attemptId, "artifact.json"));
attemptUsages.push(attempt.usage);
}
const [score, artifact] = await Promise.all([
loadObject(join(runsOut, finalAttemptId, "score.json")),
loadObject(join(runsOut, finalAttemptId, "artifact.json")),
@ -490,6 +497,8 @@ export async function aggregateProtocolEvalCampaign({
schema: "paperclip.runner-protocol-eval.campaign/v1",
campaignId: catalog.campaignId,
generatedAt,
selection: catalog.selection,
costs: sumAttemptCosts(attemptUsages),
source: {
paperclip: source.paperclip,
evals: source.evals,

View File

@ -111,6 +111,7 @@ test("catalogs roster plus case cells and emits bounded balanced shards", async
maxParallel: 80,
});
assert.equal(catalog.cells.length, 1);
assert.equal(catalog.selection.kind, "maintained_full");
assert.equal(catalog.cells[0].credentialName, "OPENROUTER_API_KEY");
assert.equal(catalog.maxParallelPerShard, 40);
assert.equal(catalog.matrices[0].include.length, 1);
@ -185,6 +186,7 @@ test("all selects the maintained enabled campaign and explicit diagnostics can s
diagnostic.rosters.map((roster) => roster.rosterId),
["protocol-live-disabled-model"],
);
assert.equal(diagnostic.selection.kind, "subset");
});
test("all fails closed when the maintained campaign is missing", async () => {
@ -252,6 +254,37 @@ test("aggregates retained attempts and synthesizes missing cells as infrastructu
);
});
test("campaign cost includes failed attempts before a successful retry", async () => {
const { root, config, evalCase } = await fixture();
const catalog = await buildProtocolEvalCatalog({ evalsRoot: root, campaignId: "gha-42-1" });
const catalogPath = join(root, "catalog.json");
const download = join(root, "downloads/cell");
await mkdir(download, { recursive: true });
await writeFile(catalogPath, JSON.stringify(catalog));
await writeFile(join(download, "cell.json"), JSON.stringify({
cellId: catalog.cells[0].cellId, caseId: evalCase.id,
rosterFile: catalog.cells[0].rosterFile, exitCode: 0,
}));
for (const [attemptId, amount, passed] of [["attempt-01", 2, false], ["attempt-02", 3, true]]) {
const directory = join(download, "runs", attemptId);
await mkdir(directory, { recursive: true });
for (const [file, value] of Object.entries({
"artifact.json": { attemptId, usage: { estimatedCostNanodollars: amount } },
"score.json": { attemptId, caseId: evalCase.id, passed, disposition: passed ? "passed" : "infrastructure_failure" },
"case.json": evalCase, "config.json": config,
})) await writeFile(join(directory, file), JSON.stringify(value));
}
const result = await aggregateProtocolEvalCampaign({
catalogPath, downloadsRoot: join(root, "downloads"), evalsRoot: root,
runsOut: join(root, "merged"), campaignOut: join(root, "campaign.json"), source: {},
});
assert.equal(result.totals.passed, 1);
assert.equal(result.results[0].usage.estimatedCostNanodollars, 3);
assert.equal(result.costs.estimated.nanodollars, 5);
assert.equal(result.costs.attempts, 2);
assert.equal(result.selection.kind, "maintained_full");
});
test("rejects downloaded cells that were not declared by the immutable catalog", async () => {
const { root } = await fixture();
const catalog = await buildProtocolEvalCatalog({

View File

@ -0,0 +1,123 @@
import { compareCampaignAnalytics } from "./runner-protocol-eval-metrics.mjs";
const escape = (value) => String(value ?? "").replaceAll("&", "&amp;").replaceAll("<", "&lt;").replaceAll(">", "&gt;").replaceAll('"', "&quot;").replaceAll("'", "&#39;");
const date = (value) => new Intl.DateTimeFormat("en-US", { dateStyle: "medium", timeStyle: "short", timeZone: "UTC" }).format(new Date(value));
const percentage = (run) => run.totals.selected ? 100 * run.totals.passed / run.totals.selected : null;
const dollars = (value) => `$${(value / 1e9).toFixed(6)}`;
export function costLabel(costs, field = "estimated") {
const metric = costs?.[field];
if (!Number.isFinite(metric?.nanodollars) || metric.nanodollars < 0) return "Unknown";
const incomplete = metric.recordedAttempts < costs.attempts;
return `${incomplete ? "≥ " : ""}${dollars(metric.nanodollars)}`;
}
function costCell(costs) {
const coverage = costs?.estimated?.recordedAttempts ?? 0;
const attempts = costs?.attempts ?? 0;
return `<strong>${escape(costLabel(costs))}</strong><small>Estimated · ${coverage}/${attempts} entries</small><small>Provider list: ${escape(costLabel(costs, "providerReported"))}</small><small>${costs?.scope === "all_attempts" ? "All attempts, including retries" : "Historical final attempts only"}</small>`;
}
function commit(repository, sha, label) {
if (!/^[a-f0-9]{40}$/.test(sha ?? "")) return `<small>${label}: unknown</small>`;
return `<small>${label}: <a href="https://github.com/${repository}/commit/${sha}" title="${sha}"><code>${sha.slice(0, 8)}</code></a></small>`;
}
function sourceLinks(run) {
const url = run.source?.workflowRunUrl;
const workflow = /^https:\/\/github\.com\/paperclipai\/paperclip\/actions\/runs\/[1-9][0-9]*$/.test(url ?? "")
? `<small><a href="${url}">GitHub Actions ↗</a></small>` : "";
return `${commit("paperclipai/paperclip", run.source?.paperclip?.sha, "Paperclip")}${commit("paperclipai/paperclip-evals", run.source?.evals?.sha, "Evals")}${workflow}<small>${escape(run.source?.paperclip?.ref ?? "")}</small>`;
}
function chart(runs, analytics, kind, id) {
const cost = kind === "cost";
const label = cost ? "Estimated cost over time (USD)" : "Pass rate over time (%)";
const values = runs.map((run) => cost ? analytics[run.campaignId]?.costs?.estimated?.nanodollars : percentage(run));
const max = cost ? Math.max(1, ...values.filter(Number.isFinite)) : 100;
const timestamps = runs.map((run) => Date.parse(run.generatedAt));
const elapsed = timestamps.at(-1) - timestamps[0];
const x = (index) => elapsed > 0 ? 48 + 504 * (timestamps[index] - timestamps[0]) / elapsed : 300;
const y = (value) => 145 - value / max * 116;
const scopes = cost ? ["all_attempts", "final_attempts"] : ["pass"];
const series = scopes.map((scope) => {
let drawing = false;
const path = values.map((value, index) => {
const present = Number.isFinite(value) && value >= 0 && (!cost || analytics[runs[index].campaignId]?.costs?.scope === scope);
if (!present) { drawing = false; return ""; }
const point = `${drawing ? "L" : "M"}${x(index).toFixed(2)},${y(value).toFixed(2)}`;
drawing = true;
return point;
}).join(" ");
return `<path class="trend-line ${scope}" d="${path}"/>`;
}).join("");
const points = values.map((value, index) => {
if (!Number.isFinite(value) || value < 0) return "";
const run = runs[index];
const metric = analytics[run.campaignId];
const description = `${date(run.generatedAt)} UTC · ${run.campaignId} · ${cost ? `${costLabel(metric?.costs)} (${metric?.costs?.scope})` : `${value.toFixed(1)}% (${run.totals.passed}/${run.totals.selected})`}`;
return `<a href="${escape(run.publicUrl)}" aria-label="${escape(description)}"><circle class="trend-point ${cost ? metric?.costs?.scope : "pass"}" cx="${x(index).toFixed(2)}" cy="${y(value).toFixed(2)}" r="4"><title>${escape(description)}</title></circle></a>`;
}).join("");
const tick = (run) => `${run.generatedAt.slice(5, 16).replace("T", " ")} UTC`;
return `<figure><figcaption>${label}</figcaption><svg class="trend-chart" viewBox="0 0 600 186" role="img" aria-labelledby="${id}"><title id="${id}">${label}. Each point links to its recorded run.</title><path class="trend-axis" d="M48,25V145H552"/><text x="40" y="33" text-anchor="end">${cost ? `$${(max / 1e9).toFixed(2)}` : "100%"}</text><text x="40" y="149" text-anchor="end">0</text>${series}${points}<text x="48" y="176">${escape(tick(runs[0]))}</text><text x="552" y="176" text-anchor="end">${escape(tick(runs.at(-1)))}</text></svg></figure>`;
}
function changeCell(run, analytics, previous) {
if (run.reportRevision) return '<span class="muted">Presentation only</span>';
if (!previous) return '<span class="muted">No matching baseline</span>';
const change = compareCampaignAnalytics(analytics[run.campaignId], analytics[previous.campaignId]);
if (!change) return '<span class="muted">Suite changed</span>';
const failures = (cells, destination) => cells.map((cell) => `<li><a href="${escape(destination.publicUrl)}tests/${encodeURIComponent(cell.caseId)}.html">${escape(cell.caseId)}</a> · ${escape(cell.rosterId)}${cell.disposition === "infrastructure_failure" ? " · infrastructure" : ""}</li>`).join("");
const counts = `${change.regressions.length} regressions · ${change.recoveries.length} recoveries`;
return `<details><summary>${counts}</summary><small>Versus <a href="${escape(previous.publicUrl)}">${escape(previous.campaignId)}</a></small>${change.regressions.length ? `<strong>Previously passing → failing</strong><ul>${failures(change.regressions, run)}</ul>` : ""}${change.recoveries.length ? `<strong>Previously failing → passing</strong><ul>${failures(change.recoveries, run)}</ul>` : ""}</details>`;
}
export function renderProtocolEvalHistoryIndex(history, stylesheetHref) {
if (!/^campaigns\/gha-[a-z0-9-]+\/viewer\/assets\/[A-Za-z0-9._-]+\.css$/.test(stylesheetHref ?? ""))
throw new Error("History requires an immutable campaign's Runner Lab stylesheet");
for (const run of history.campaigns) {
if (!["selected", "passed", "behaviorFailures", "infrastructureFailures"].every((field) => Number.isSafeInteger(run.totals?.[field]) && run.totals[field] >= 0)
|| !run.rosters.every((roster) => Number.isSafeInteger(roster.selected) && Number.isSafeInteger(roster.passed)))
throw new Error("History requires numeric recorded counts");
const url = new URL(run.publicUrl);
if (url.protocol !== "https:" || url.username || url.password)
throw new Error("History report links must be credential-free HTTPS URLs");
}
const analytics = history.analytics ?? {};
const presentations = new Map();
for (const refresh of history.campaigns.filter((run) => run.reportRevision)
.sort((a, b) => a.reportRevision.renderedAt.localeCompare(b.reportRevision.renderedAt)))
presentations.set(refresh.reportRevision.sourceCampaignId, refresh);
const measurements = history.campaigns.filter((run) => !run.reportRevision)
.map((run) => presentations.has(run.campaignId)
? { ...run, originalPublicUrl: run.publicUrl, publicUrl: presentations.get(run.campaignId).publicUrl }
: run)
.sort((a, b) => a.generatedAt.localeCompare(b.generatedAt) || a.campaignId.localeCompare(b.campaignId));
const groups = new Map();
const baselines = new Map();
for (const run of measurements) {
const key = analytics[run.campaignId]?.suiteKey;
if (!key || !run.complete) continue;
const group = groups.get(key) ?? [];
baselines.set(run.campaignId, group.at(-1));
group.push(run);
groups.set(key, group);
}
const ordered = [...groups.entries()].sort(([, a], [, b]) => b.at(-1).totals.selected - a.at(-1).totals.selected || b.at(-1).generatedAt.localeCompare(a.at(-1).generatedAt));
const trends = ordered.map(([key, runs], index) => {
const latest = runs.at(-1);
return `<details class="trend-group" ${index === 0 ? "open" : ""}><summary>${latest.totals.selected} cells · ${latest.rosters.length} configurations · ${runs.length} recorded runs · evals ${escape(latest.source.evals.sha.slice(0, 8))}</summary><div class="trend-grid">${chart(runs, analytics, "pass", `pass-${key}`)}${chart(runs, analytics, "cost", `cost-${key}`)}</div><p class="muted">Cost: solid cyan includes retries; dashed purple is historical final-attempt cost. Missing cost is a gap, not zero. Hover or select a point to inspect its run.</p></details>`;
}).join("") || '<p class="muted">Comparable run metadata has not been recorded yet.</p>';
const row = (run) => {
const metric = analytics[run.reportRevision?.sourceCampaignId ?? run.campaignId];
const status = !run.complete ? "incomplete" : run.allPassed ? "passed" : "failed";
const models = run.rosters.map((roster) => `<li>${escape(roster.model)} · ${roster.passed}/${roster.selected}<small>${escape(roster.driver)} · ${escape(roster.rosterId)}</small></li>`).join("");
const scope = metric?.selection === "maintained_full" ? "Full maintained suite" : metric?.selection === "subset" ? "Selected subset" : `${run.totals.selected} recorded cells`;
return `<tr><td><a href="${escape(run.publicUrl)}"><code>${escape(run.campaignId)}</code></a><small>${escape(date(run.generatedAt))} UTC</small>${run.reportRevision ? `<small>Report refresh · no new model calls · source ${escape(run.reportRevision.sourceCampaignId)}</small>` : `<small>${scope}</small>`}<small><a href="${escape(run.publicUrl)}">Open Evalbook →</a></small></td><td><span class="status ${status}">${status}</span><strong class="pass-rate">${run.totals.passed}/${run.totals.selected} · ${percentage(run)?.toFixed(1) ?? "—"}%</strong><small>${run.totals.behaviorFailures} behavior · ${run.totals.infrastructureFailures} infrastructure</small><details><summary>${run.rosters.length} model configurations</summary><ul>${models}</ul></details></td><td>${changeCell(run, analytics, baselines.get(run.campaignId))}</td><td>${run.reportRevision ? '<small>No additional model cost</small>' : costCell(metric?.costs)}</td><td>${sourceLinks(run)}</td></tr>`;
};
const table = (runs) => `<div class="table history-table" role="region" aria-label="Recorded eval runs" tabindex="0"><table><thead><tr><th>Run</th><th>Pass / fail</th><th>Change vs matching suite</th><th>Cost (USD)</th><th>Exact source</th></tr></thead><tbody>${runs.map(row).join("") || '<tr><td colspan="5">No campaigns have been published yet.</td></tr>'}</tbody></table></div>`;
const refreshes = history.campaigns.filter((run) => run.reportRevision);
const latest = measurements.find((run) => run.campaignId === history.latestCampaignId);
const green = measurements.find((run) => run.campaignId === history.latestGreenCampaignId);
return `<!doctype html><html lang="en"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width"><title>Runner protocol eval campaigns · Paperclip</title><link rel="stylesheet" href="${escape(stylesheetHref)}"></head><body class="evalbook-site"><main><header class="top"><a href="index.html">paperclip-runner evals</a><span class="badge">Run history</span></header><h1>Runner protocol eval campaigns</h1><p class="muted">Every recorded campaign, its cost, and its exact source. Public chat replays are linked below; full provider evidence remains in access-controlled workflow artifacts.</p><nav class="pointers">${latest ? `<a href="${escape(latest.publicUrl)}">Latest · ${escape(latest.campaignId)}</a>` : ""}${green ? `<a href="${escape(green.publicUrl)}">Latest green · ${escape(green.campaignId)}</a>` : ""}<a href="history.json">Download history JSON</a></nav><h2>Like-for-like trends</h2><p class="muted">Only identical cells, model configurations, and eval-suite SHAs are compared. A changed suite starts a separate series. Report refreshes never count as new measurements. Regressions distinguish model behavior from infrastructure failures.</p>${trends}<h2>All runs · ${measurements.length}</h2><p class="muted">Estimated cost and provider list cost are alternatives, not additive. ≥ means some entries lack usage. Historical final-only cost excludes retry spending. Commit links expose the full SHA on hover.</p>${table([...measurements].reverse())}${refreshes.length ? `<details><summary>Report refreshes · ${refreshes.length} (no new measurements)</summary>${table(refreshes)}</details>` : ""}<footer>Updated ${escape(date(history.updatedAt))} UTC · All run records retained · Immutable campaign bundles</footer></main></body></html>`;
}

View File

@ -0,0 +1,97 @@
import { createHash } from "node:crypto";
import { isDeepStrictEqual } from "node:util";
const SHA = /^[a-f0-9]{40}$/;
const ID = /^[A-Za-z0-9][A-Za-z0-9._-]{0,199}$/;
// Provider list cost and our pricing estimate are alternative measurements,
// never additive. Missing usage is unknown, not a zero-dollar call.
export function sumAttemptCosts(usages, scope = "all_attempts") {
const metric = (field) => {
const values = usages.map((usage) => usage?.[field]).filter((value) =>
typeof value === "number" && Number.isFinite(value) && value >= 0);
return {
nanodollars: values.length ? values.reduce((sum, value) => sum + value, 0) : null,
recordedAttempts: values.length,
};
};
return {
scope,
attempts: usages.length,
estimated: metric("estimatedCostNanodollars"),
providerReported: metric("providerReportedCostNanodollars"),
};
}
export function campaignAnalytics(campaign) {
const results = campaign.results ?? [];
const cells = results.map((result) => [result.cellId, result.model, result.provider, result.driver]);
cells.sort((a, b) => JSON.stringify(a).localeCompare(JSON.stringify(b)));
const comparable = SHA.test(campaign.source?.evals?.sha ?? "")
&& cells.length > 0 && cells.length === campaign.totals?.selected
&& results.every((result) => typeof result.passed === "boolean")
&& results.filter((result) => result.passed).length === campaign.totals.passed
&& new Set(cells.map(([id]) => id)).size === cells.length
&& cells.every((cell) => cell.every((value) => typeof value === "string" && value.length > 0));
const suiteKey = comparable ? createHash("sha256")
.update(JSON.stringify([campaign.source.evals.sha, cells])).digest("hex") : null;
const suppliedCosts = campaign.costs;
const validCosts = suppliedCosts && ["all_attempts", "final_attempts"].includes(suppliedCosts.scope)
&& Number.isSafeInteger(suppliedCosts.attempts) && suppliedCosts.attempts >= 0
&& ["estimated", "providerReported"].every((field) => {
const metric = suppliedCosts[field];
return Number.isSafeInteger(metric?.recordedAttempts) && metric.recordedAttempts >= 0
&& metric.recordedAttempts <= suppliedCosts.attempts
&& (metric.nanodollars === null || typeof metric.nanodollars === "number" && Number.isFinite(metric.nanodollars) && metric.nanodollars >= 0);
});
return {
schema: "paperclip.runner-protocol-eval.analytics/v1",
suiteKey,
selection: campaign.selection?.kind === "maintained_full" ? "maintained_full"
: campaign.selection?.kind === "subset" ? "subset" : "recorded",
costs: validCosts ? {
scope: suppliedCosts.scope, attempts: suppliedCosts.attempts,
estimated: { nanodollars: suppliedCosts.estimated.nanodollars, recordedAttempts: suppliedCosts.estimated.recordedAttempts },
providerReported: { nanodollars: suppliedCosts.providerReported.nanodollars, recordedAttempts: suppliedCosts.providerReported.recordedAttempts },
} : sumAttemptCosts(results.map((result) => result.usage), "final_attempts"),
failedCells: results.filter((result) => !result.passed).map((result) => ({
cellId: ID.test(result.cellId) ? result.cellId : "unknown",
caseId: ID.test(result.caseId) ? result.caseId : "unknown",
rosterId: ID.test(result.rosterId) ? result.rosterId : "unknown",
disposition: result.disposition === "behavior_failure" ? "behavior_failure" : "infrastructure_failure",
})),
};
}
export function compareCampaignAnalytics(current, previous) {
if (!current?.suiteKey || current.suiteKey !== previous?.suiteKey) return null;
const before = new Set(previous.failedCells.map((cell) => cell.cellId));
const after = new Set(current.failedCells.map((cell) => cell.cellId));
return {
regressions: current.failedCells.filter((cell) => !before.has(cell.cellId)),
recoveries: previous.failedCells.filter((cell) => !after.has(cell.cellId)),
};
}
// History records and campaign bundles stay immutable. This separate, derived
// projection can be backfilled from old bundles or enriched by a report refresh.
export async function enrichProtocolEvalHistory(history, { currentCampaign, loadCampaign }) {
const analytics = { ...history.analytics };
const sourceId = currentCampaign.reportRevision?.sourceCampaignId ?? currentCampaign.campaignId;
for (const record of history.campaigns.filter((item) => !item.reportRevision)) {
const refresh = record.campaignId === sourceId ? currentCampaign : null;
if (!refresh && analytics[record.campaignId]?.schema === "paperclip.runner-protocol-eval.analytics/v1") continue;
const campaign = refresh ?? await loadCampaign(record.campaignId);
if (!campaign) continue;
if (!isDeepStrictEqual(campaign.source, record.source)
|| !isDeepStrictEqual(campaign.totals, record.totals))
throw new Error(`Campaign analytics do not match immutable history: ${record.campaignId}`);
const projected = campaignAnalytics(campaign);
const previous = analytics[record.campaignId];
// Never downgrade an all-attempt measurement with a historical final-only one.
if (previous?.costs?.scope === "all_attempts" && projected.costs.scope !== "all_attempts")
projected.costs = previous.costs;
analytics[record.campaignId] = projected;
}
return { ...history, analytics };
}

View File

@ -0,0 +1,83 @@
import assert from "node:assert/strict";
import test from "node:test";
import { campaignAnalytics, compareCampaignAnalytics, enrichProtocolEvalHistory, sumAttemptCosts } from "./runner-protocol-eval-metrics.mjs";
import { costLabel, renderProtocolEvalHistoryIndex } from "./runner-protocol-eval-history-view.mjs";
import { emptyProtocolEvalHistory, mergeProtocolEvalHistory, protocolEvalHistoryRecord } from "./publish-runner-protocol-eval-history.mjs";
function campaign(id, failed = [], overrides = {}) {
const results = ["context", "create", "finish"].map((caseId) => ({
cellId: `codex--${caseId}`, caseId, rosterId: "codex", model: "mini", provider: "codex", driver: "codex_app_server",
passed: !failed.includes(caseId), disposition: failed.includes(caseId) ? "behavior_failure" : "passed",
usage: { estimatedCostNanodollars: 1_000_000, providerReportedCostNanodollars: 2_000_000 },
}));
return {
campaignId: `gha-${id}-1`, generatedAt: `2026-09-${String(id).padStart(2, "0")}T00:00:00.000Z`,
source: { paperclip: { sha: "a".repeat(40), ref: "refs/heads/master" }, evals: { sha: "b".repeat(40) }, workflowRunUrl: `https://github.com/paperclipai/paperclip/actions/runs/${id}` },
complete: true, allPassed: failed.length === 0, results,
selection: { kind: "maintained_full" },
totals: { selected: 3, passed: 3 - failed.length, behaviorFailures: failed.length, infrastructureFailures: 0 },
rosters: [{ rosterId: "codex", model: "mini", driver: "codex_app_server", selected: 3, passed: 3 - failed.length }],
...overrides,
};
}
test("costs include retries without adding provider list cost to estimates or inventing missing usage", () => {
const costs = sumAttemptCosts([{ estimatedCostNanodollars: 2, providerReportedCostNanodollars: 7 }, { estimatedCostNanodollars: 3 }, {}, { estimatedCostNanodollars: -1 }, { estimatedCostNanodollars: Infinity }]);
assert.equal(costs.scope, "all_attempts");
assert.deepEqual(costs.estimated, { nanodollars: 5, recordedAttempts: 2 });
assert.deepEqual(costs.providerReported, { nanodollars: 7, recordedAttempts: 1 });
assert.equal(costs.attempts, 5);
assert.match(costLabel(costs), /^≥ /);
assert.equal(costLabel(sumAttemptCosts([{}, {}])), "Unknown");
assert.equal(costLabel(sumAttemptCosts([{ estimatedCostNanodollars: 0 }])), "$0.000000");
});
test("regressions and recoveries compare exact suite membership and evals SHA, not count alone", () => {
const before = campaignAnalytics(campaign(1, ["context"]));
const after = campaignAnalytics(campaign(2, ["create"]));
const change = compareCampaignAnalytics(after, before);
assert.deepEqual(change.regressions.map((cell) => cell.caseId), ["create"]);
assert.deepEqual(change.recoveries.map((cell) => cell.caseId), ["context"]);
const differentSha = campaign(2, [], { source: { evals: { sha: "c".repeat(40) } } });
assert.equal(compareCampaignAnalytics(campaignAnalytics(differentSha), before), null);
const changedMembership = campaign(2);
changedMembership.results[0].cellId = "codex--different-case";
assert.equal(compareCampaignAnalytics(campaignAnalytics(changedMembership), before), null);
const reordered = campaign(1, ["context"]);
reordered.results.reverse();
assert.equal(campaignAnalytics(reordered).suiteKey, before.suiteKey);
});
test("history analytics backfill without rewriting records and refreshes enrich original costs", async () => {
const source = campaign(1);
const record = protocolEvalHistoryRecord(source, "https://reports.example/runner-protocol-evals");
let history = mergeProtocolEvalHistory(emptyProtocolEvalHistory(), record);
history = await enrichProtocolEvalHistory(history, { currentCampaign: source, loadCampaign: () => assert.fail("current campaign already available") });
assert.equal(history.analytics[source.campaignId].costs.scope, "final_attempts");
const refresh = { ...source, campaignId: "gha-1-1-report-theme", reportRevision: { sourceCampaignId: source.campaignId, renderedAt: "2026-09-04T00:00:00.000Z" }, costs: sumAttemptCosts([...source.results.map((r) => r.usage), { estimatedCostNanodollars: 500_000 }]) };
history = mergeProtocolEvalHistory(history, protocolEvalHistoryRecord(refresh, "https://reports.example/runner-protocol-evals"));
history = await enrichProtocolEvalHistory(history, { currentCampaign: refresh, loadCampaign: () => assert.fail("refresh already available") });
assert.equal(history.analytics[source.campaignId].costs.estimated.nanodollars, 3_500_000);
assert.deepEqual(history.campaigns.find((run) => run.campaignId === source.campaignId), record);
assert.equal(history.latestCampaignId, source.campaignId);
await assert.rejects(enrichProtocolEvalHistory(history, { currentCampaign: { ...refresh, totals: { ...source.totals, passed: 0 } }, loadCampaign: () => null }), /do not match/);
});
test("history graphs exclude refreshes, link exact SHAs and Actions, and open the latest presentation", async () => {
const first = campaign(1, ["context"]);
const second = campaign(2, ["create"]);
const refresh = { ...second, campaignId: "gha-2-1-report-theme", reportRevision: { sourceCampaignId: second.campaignId, renderedAt: "2026-09-04T00:00:00.000Z" } };
let history = emptyProtocolEvalHistory();
for (const item of [first, second, refresh]) history = mergeProtocolEvalHistory(history, protocolEvalHistoryRecord(item, "https://reports.example/runner-protocol-evals"));
history = await enrichProtocolEvalHistory(history, { currentCampaign: refresh, loadCampaign: (id) => id === first.campaignId ? first : second });
const html = renderProtocolEvalHistoryIndex(history, "campaigns/gha-2-1-report-theme/viewer/assets/index.css");
assert.match(html, /All runs · 2/);
assert.match(html, /2 recorded runs/);
assert.match(html, /1 regressions · 1 recoveries/);
assert.match(html, /Full maintained suite/);
assert.ok(html.includes(`https://github.com/paperclipai/paperclip/commit/${"a".repeat(40)}`));
assert.ok(html.includes(`https://github.com/paperclipai/paperclip-evals/commit/${"b".repeat(40)}`));
assert.match(html, /actions\/runs\/2/);
assert.match(html, /href="https:\/\/reports.example\/runner-protocol-evals\/campaigns\/gha-2-1-report-theme\/">Latest · gha-2-1/);
assert.doesNotMatch(html, /<script|NaN|Infinity/);
});