mirror of https://github.com/garrytan/gstack.git
Merge a7f50af250 into a3259400a3
This commit is contained in:
commit
0f2f2f26fc
|
|
@ -2,11 +2,11 @@
|
||||||
name: benchmark
|
name: benchmark
|
||||||
preamble-tier: 1
|
preamble-tier: 1
|
||||||
version: 1.0.0
|
version: 1.0.0
|
||||||
description: Performance regression detection using the browse daemon. (gstack)
|
description: Web page performance regression detection using the browse daemon. (gstack)
|
||||||
triggers:
|
triggers:
|
||||||
- performance benchmark
|
- web performance benchmark
|
||||||
- check page speed
|
- check page speed
|
||||||
- detect performance regression
|
- detect page performance regression
|
||||||
allowed-tools:
|
allowed-tools:
|
||||||
- Bash
|
- Bash
|
||||||
- Read
|
- Read
|
||||||
|
|
@ -23,8 +23,9 @@ allowed-tools:
|
||||||
Establishes
|
Establishes
|
||||||
baselines for page load times, Core Web Vitals, and resource sizes.
|
baselines for page load times, Core Web Vitals, and resource sizes.
|
||||||
Compares before/after on every PR. Tracks performance trends over time.
|
Compares before/after on every PR. Tracks performance trends over time.
|
||||||
Use when: "performance", "benchmark", "page speed", "lighthouse", "web vitals",
|
Use when: "web performance benchmark", "page speed", "lighthouse",
|
||||||
"bundle size", "load time".
|
"web vitals", "bundle size", "load time". Do not use for AI model, product,
|
||||||
|
business, eval, or research benchmark work.
|
||||||
|
|
||||||
Voice triggers (speech-to-text aliases): "speed test", "check performance".
|
Voice triggers (speech-to-text aliases): "speed test", "check performance".
|
||||||
|
|
||||||
|
|
@ -577,6 +578,13 @@ You are a **Performance Engineer** who has optimized apps serving millions of re
|
||||||
|
|
||||||
Your job is to measure, baseline, compare, and alert. You use the browse daemon's `perf` command and JavaScript evaluation to gather real performance data from running pages.
|
Your job is to measure, baseline, compare, and alert. You use the browse daemon's `perf` command and JavaScript evaluation to gather real performance data from running pages.
|
||||||
|
|
||||||
|
## Not For
|
||||||
|
|
||||||
|
Do not run this skill for the bare word "benchmark" unless the request is clearly
|
||||||
|
about web page performance. Skip it for model benchmarks, agent evals, business
|
||||||
|
benchmark research, product metric design, or atrib behavior measurement. For
|
||||||
|
cross-model gstack skill comparisons, use `/benchmark-models`.
|
||||||
|
|
||||||
## User-invocable
|
## User-invocable
|
||||||
When the user types `/benchmark`, run this skill.
|
When the user types `/benchmark`, run this skill.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -3,18 +3,19 @@ name: benchmark
|
||||||
preamble-tier: 1
|
preamble-tier: 1
|
||||||
version: 1.0.0
|
version: 1.0.0
|
||||||
description: |
|
description: |
|
||||||
Performance regression detection using the browse daemon. Establishes
|
Web page performance regression detection using the browse daemon. Establishes
|
||||||
baselines for page load times, Core Web Vitals, and resource sizes.
|
baselines for page load times, Core Web Vitals, and resource sizes.
|
||||||
Compares before/after on every PR. Tracks performance trends over time.
|
Compares before/after on every PR. Tracks performance trends over time.
|
||||||
Use when: "performance", "benchmark", "page speed", "lighthouse", "web vitals",
|
Use when: "web performance benchmark", "page speed", "lighthouse",
|
||||||
"bundle size", "load time". (gstack)
|
"web vitals", "bundle size", "load time". Do not use for AI model, product,
|
||||||
|
business, eval, or research benchmark work. (gstack)
|
||||||
voice-triggers:
|
voice-triggers:
|
||||||
- "speed test"
|
- "speed test"
|
||||||
- "check performance"
|
- "check performance"
|
||||||
triggers:
|
triggers:
|
||||||
- performance benchmark
|
- web performance benchmark
|
||||||
- check page speed
|
- check page speed
|
||||||
- detect performance regression
|
- detect page performance regression
|
||||||
allowed-tools:
|
allowed-tools:
|
||||||
- Bash
|
- Bash
|
||||||
- Read
|
- Read
|
||||||
|
|
@ -33,6 +34,13 @@ You are a **Performance Engineer** who has optimized apps serving millions of re
|
||||||
|
|
||||||
Your job is to measure, baseline, compare, and alert. You use the browse daemon's `perf` command and JavaScript evaluation to gather real performance data from running pages.
|
Your job is to measure, baseline, compare, and alert. You use the browse daemon's `perf` command and JavaScript evaluation to gather real performance data from running pages.
|
||||||
|
|
||||||
|
## Not For
|
||||||
|
|
||||||
|
Do not run this skill for the bare word "benchmark" unless the request is clearly
|
||||||
|
about web page performance. Skip it for model benchmarks, agent evals, business
|
||||||
|
benchmark research, product metric design, or atrib behavior measurement. For
|
||||||
|
cross-model gstack skill comparisons, use `/benchmark-models`.
|
||||||
|
|
||||||
## User-invocable
|
## User-invocable
|
||||||
When the user types `/benchmark`, run this skill.
|
When the user types `/benchmark`, run this skill.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -11,7 +11,7 @@ Conventions:
|
||||||
## Skills
|
## Skills
|
||||||
|
|
||||||
- [/autoplan](autoplan/SKILL.md): Auto-review pipeline — reads the full CEO, design, eng, and DX review skills from disk and runs them sequentially with auto-decisions using 6 decision principles.
|
- [/autoplan](autoplan/SKILL.md): Auto-review pipeline — reads the full CEO, design, eng, and DX review skills from disk and runs them sequentially with auto-decisions using 6 decision principles.
|
||||||
- [/benchmark](benchmark/SKILL.md): Performance regression detection using the browse daemon.
|
- [/benchmark](benchmark/SKILL.md): Web page performance regression detection using the browse daemon.
|
||||||
- [/benchmark-models](benchmark-models/SKILL.md): Cross-model benchmark for gstack skills.
|
- [/benchmark-models](benchmark-models/SKILL.md): Cross-model benchmark for gstack skills.
|
||||||
- [/browse](browse/SKILL.md): Fast headless browser for QA testing and site dogfooding.
|
- [/browse](browse/SKILL.md): Fast headless browser for QA testing and site dogfooding.
|
||||||
- [/canary](canary/SKILL.md): Post-deploy canary monitoring.
|
- [/canary](canary/SKILL.md): Post-deploy canary monitoring.
|
||||||
|
|
|
||||||
|
|
@ -9,8 +9,8 @@
|
||||||
"voice_line": "Voice triggers (speech-to-text aliases): \"auto plan\", \"automatic review\"."
|
"voice_line": "Voice triggers (speech-to-text aliases): \"auto plan\", \"automatic review\"."
|
||||||
},
|
},
|
||||||
"benchmark": {
|
"benchmark": {
|
||||||
"lead": "Performance regression detection using the browse daemon.",
|
"lead": "Web page performance regression detection using the browse daemon.",
|
||||||
"routing": "Establishes\nbaselines for page load times, Core Web Vitals, and resource sizes.\nCompares before/after on every PR. Tracks performance trends over time.\nUse when: \"performance\", \"benchmark\", \"page speed\", \"lighthouse\", \"web vitals\",\n\"bundle size\", \"load time\".",
|
"routing": "Establishes\nbaselines for page load times, Core Web Vitals, and resource sizes.\nCompares before/after on every PR. Tracks performance trends over time.\nUse when: \"web performance benchmark\", \"page speed\", \"lighthouse\",\n\"web vitals\", \"bundle size\", \"load time\". Do not use for AI model, product,\nbusiness, eval, or research benchmark work.",
|
||||||
"voice_line": "Voice triggers (speech-to-text aliases): \"speed test\", \"check performance\"."
|
"voice_line": "Voice triggers (speech-to-text aliases): \"speed test\", \"check performance\"."
|
||||||
},
|
},
|
||||||
"benchmark-models": {
|
"benchmark-models": {
|
||||||
|
|
|
||||||
|
|
@ -467,6 +467,21 @@ describe('gen-skill-docs', () => {
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test('benchmark skill is scoped to web page performance', () => {
|
||||||
|
const generated = fs.readFileSync(path.join(ROOT, 'benchmark', 'SKILL.md'), 'utf-8');
|
||||||
|
const template = fs.readFileSync(path.join(ROOT, 'benchmark', 'SKILL.md.tmpl'), 'utf-8');
|
||||||
|
|
||||||
|
for (const content of [generated, template]) {
|
||||||
|
const frontmatter = content.match(/^---\n([\s\S]*?)\n---/)?.[1] ?? '';
|
||||||
|
expect(frontmatter).toContain('name: benchmark');
|
||||||
|
expect(frontmatter).toContain('web performance benchmark');
|
||||||
|
expect(frontmatter).not.toMatch(/Use when: [^\n]*["']benchmark["']/);
|
||||||
|
expect(content).toContain('Do not run this skill for the bare word "benchmark"');
|
||||||
|
expect(content).toContain('For\ncross-model gstack skill comparisons, use `/benchmark-models`.');
|
||||||
|
expect(content).toContain('/benchmark <url>');
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
test('qa and qa-only templates use QA_METHODOLOGY placeholder', () => {
|
test('qa and qa-only templates use QA_METHODOLOGY placeholder', () => {
|
||||||
const qaTmpl = fs.readFileSync(path.join(ROOT, 'qa', 'SKILL.md.tmpl'), 'utf-8');
|
const qaTmpl = fs.readFileSync(path.join(ROOT, 'qa', 'SKILL.md.tmpl'), 'utf-8');
|
||||||
expect(qaTmpl).toContain('{{QA_METHODOLOGY}}');
|
expect(qaTmpl).toContain('{{QA_METHODOLOGY}}');
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue