add dogfood skill for agent-driven exploratory qa (#538)
* dogfood skill * evals * haiku * fixes * caching * fixes * don't use npx
This commit is contained in:
@@ -27,10 +27,15 @@ npm-debug.log*
|
||||
.DS_Store
|
||||
Thumbs.db
|
||||
|
||||
# Python
|
||||
__pycache__/
|
||||
|
||||
# Test artifacts
|
||||
*.png
|
||||
*.jpeg
|
||||
*.jpg
|
||||
*.webm
|
||||
test/e2e/.dogfood-output/
|
||||
|
||||
# Package manager
|
||||
package-lock.json
|
||||
|
||||
@@ -32,6 +32,7 @@
|
||||
"format:check": "prettier --check 'src/**/*.ts'",
|
||||
"test": "vitest run",
|
||||
"test:watch": "vitest",
|
||||
"test:e2e:dogfood": "vitest run test/e2e/dogfood.eval.ts",
|
||||
"postinstall": "node scripts/postinstall.js",
|
||||
"changeset": "changeset",
|
||||
"ci:version": "changeset version && pnpm run version:sync && pnpm install --no-frozen-lockfile",
|
||||
@@ -62,6 +63,7 @@
|
||||
"zod": "^3.22.4"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@anthropic-ai/claude-agent-sdk": "^0.2.52",
|
||||
"@changesets/cli": "^2.29.8",
|
||||
"@types/node": "^20.10.0",
|
||||
"@types/ws": "^8.18.1",
|
||||
|
||||
Generated
+174
@@ -24,6 +24,9 @@ importers:
|
||||
specifier: ^3.22.4
|
||||
version: 3.25.76
|
||||
devDependencies:
|
||||
'@anthropic-ai/claude-agent-sdk':
|
||||
specifier: ^0.2.52
|
||||
version: 0.2.52(zod@3.25.76)
|
||||
'@changesets/cli':
|
||||
specifier: ^2.29.8
|
||||
version: 2.29.8(@types/node@20.19.28)
|
||||
@@ -57,6 +60,12 @@ importers:
|
||||
|
||||
packages:
|
||||
|
||||
'@anthropic-ai/claude-agent-sdk@0.2.52':
|
||||
resolution: {integrity: sha512-rdTQUu/HjKlDNNxJuhtXY6LJDOLvzVBU7sXFuFIG6CEC/nFfcvYq035EyjVw4nzu7lLZim/m+g2yZ8uNIcbaFw==}
|
||||
engines: {node: '>=18.0.0'}
|
||||
peerDependencies:
|
||||
zod: ^4.0.0
|
||||
|
||||
'@appium/logger@1.7.1':
|
||||
resolution: {integrity: sha512-9C2o9X/lBEDBUnKfAi3mRo9oG7Z03nmISLwsGkWxIWjMAvBdJD0RRSJMekWVKzfXN3byrI1WlCXTITzN4LAoLw==}
|
||||
engines: {node: ^14.17.0 || ^16.13.0 || >=18.0.0, npm: '>=8'}
|
||||
@@ -276,6 +285,95 @@ packages:
|
||||
cpu: [x64]
|
||||
os: [win32]
|
||||
|
||||
'@img/sharp-darwin-arm64@0.34.5':
|
||||
resolution: {integrity: sha512-imtQ3WMJXbMY4fxb/Ndp6HBTNVtWCUI0WdobyheGf5+ad6xX8VIDO8u2xE4qc/fr08CKG/7dDseFtn6M6g/r3w==}
|
||||
engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0}
|
||||
cpu: [arm64]
|
||||
os: [darwin]
|
||||
|
||||
'@img/sharp-darwin-x64@0.34.5':
|
||||
resolution: {integrity: sha512-YNEFAF/4KQ/PeW0N+r+aVVsoIY0/qxxikF2SWdp+NRkmMB7y9LBZAVqQ4yhGCm/H3H270OSykqmQMKLBhBJDEw==}
|
||||
engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0}
|
||||
cpu: [x64]
|
||||
os: [darwin]
|
||||
|
||||
'@img/sharp-libvips-darwin-arm64@1.2.4':
|
||||
resolution: {integrity: sha512-zqjjo7RatFfFoP0MkQ51jfuFZBnVE2pRiaydKJ1G/rHZvnsrHAOcQALIi9sA5co5xenQdTugCvtb1cuf78Vf4g==}
|
||||
cpu: [arm64]
|
||||
os: [darwin]
|
||||
|
||||
'@img/sharp-libvips-darwin-x64@1.2.4':
|
||||
resolution: {integrity: sha512-1IOd5xfVhlGwX+zXv2N93k0yMONvUlANylbJw1eTah8K/Jtpi15KC+WSiaX/nBmbm2HxRM1gZ0nSdjSsrZbGKg==}
|
||||
cpu: [x64]
|
||||
os: [darwin]
|
||||
|
||||
'@img/sharp-libvips-linux-arm64@1.2.4':
|
||||
resolution: {integrity: sha512-excjX8DfsIcJ10x1Kzr4RcWe1edC9PquDRRPx3YVCvQv+U5p7Yin2s32ftzikXojb1PIFc/9Mt28/y+iRklkrw==}
|
||||
cpu: [arm64]
|
||||
os: [linux]
|
||||
|
||||
'@img/sharp-libvips-linux-arm@1.2.4':
|
||||
resolution: {integrity: sha512-bFI7xcKFELdiNCVov8e44Ia4u2byA+l3XtsAj+Q8tfCwO6BQ8iDojYdvoPMqsKDkuoOo+X6HZA0s0q11ANMQ8A==}
|
||||
cpu: [arm]
|
||||
os: [linux]
|
||||
|
||||
'@img/sharp-libvips-linux-x64@1.2.4':
|
||||
resolution: {integrity: sha512-tJxiiLsmHc9Ax1bz3oaOYBURTXGIRDODBqhveVHonrHJ9/+k89qbLl0bcJns+e4t4rvaNBxaEZsFtSfAdquPrw==}
|
||||
cpu: [x64]
|
||||
os: [linux]
|
||||
|
||||
'@img/sharp-libvips-linuxmusl-arm64@1.2.4':
|
||||
resolution: {integrity: sha512-FVQHuwx1IIuNow9QAbYUzJ+En8KcVm9Lk5+uGUQJHaZmMECZmOlix9HnH7n1TRkXMS0pGxIJokIVB9SuqZGGXw==}
|
||||
cpu: [arm64]
|
||||
os: [linux]
|
||||
|
||||
'@img/sharp-libvips-linuxmusl-x64@1.2.4':
|
||||
resolution: {integrity: sha512-+LpyBk7L44ZIXwz/VYfglaX/okxezESc6UxDSoyo2Ks6Jxc4Y7sGjpgU9s4PMgqgjj1gZCylTieNamqA1MF7Dg==}
|
||||
cpu: [x64]
|
||||
os: [linux]
|
||||
|
||||
'@img/sharp-linux-arm64@0.34.5':
|
||||
resolution: {integrity: sha512-bKQzaJRY/bkPOXyKx5EVup7qkaojECG6NLYswgktOZjaXecSAeCWiZwwiFf3/Y+O1HrauiE3FVsGxFg8c24rZg==}
|
||||
engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0}
|
||||
cpu: [arm64]
|
||||
os: [linux]
|
||||
|
||||
'@img/sharp-linux-arm@0.34.5':
|
||||
resolution: {integrity: sha512-9dLqsvwtg1uuXBGZKsxem9595+ujv0sJ6Vi8wcTANSFpwV/GONat5eCkzQo/1O6zRIkh0m/8+5BjrRr7jDUSZw==}
|
||||
engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0}
|
||||
cpu: [arm]
|
||||
os: [linux]
|
||||
|
||||
'@img/sharp-linux-x64@0.34.5':
|
||||
resolution: {integrity: sha512-MEzd8HPKxVxVenwAa+JRPwEC7QFjoPWuS5NZnBt6B3pu7EG2Ge0id1oLHZpPJdn3OQK+BQDiw9zStiHBTJQQQQ==}
|
||||
engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0}
|
||||
cpu: [x64]
|
||||
os: [linux]
|
||||
|
||||
'@img/sharp-linuxmusl-arm64@0.34.5':
|
||||
resolution: {integrity: sha512-fprJR6GtRsMt6Kyfq44IsChVZeGN97gTD331weR1ex1c1rypDEABN6Tm2xa1wE6lYb5DdEnk03NZPqA7Id21yg==}
|
||||
engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0}
|
||||
cpu: [arm64]
|
||||
os: [linux]
|
||||
|
||||
'@img/sharp-linuxmusl-x64@0.34.5':
|
||||
resolution: {integrity: sha512-Jg8wNT1MUzIvhBFxViqrEhWDGzqymo3sV7z7ZsaWbZNDLXRJZoRGrjulp60YYtV4wfY8VIKcWidjojlLcWrd8Q==}
|
||||
engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0}
|
||||
cpu: [x64]
|
||||
os: [linux]
|
||||
|
||||
'@img/sharp-win32-arm64@0.34.5':
|
||||
resolution: {integrity: sha512-WQ3AgWCWYSb2yt+IG8mnC6Jdk9Whs7O0gxphblsLvdhSpSTtmu69ZG1Gkb6NuvxsNACwiPV6cNSZNzt0KPsw7g==}
|
||||
engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0}
|
||||
cpu: [arm64]
|
||||
os: [win32]
|
||||
|
||||
'@img/sharp-win32-x64@0.34.5':
|
||||
resolution: {integrity: sha512-+29YMsqY2/9eFEiW93eqWnuLcWcufowXewwSNIT6UwZdUUCrM3oFjMWH/Z6/TMmb4hlFenmfAVbpWeup2jryCw==}
|
||||
engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0}
|
||||
cpu: [x64]
|
||||
os: [win32]
|
||||
|
||||
'@inquirer/external-editor@1.0.3':
|
||||
resolution: {integrity: sha512-RWbSrDiYmO4LbejWY7ttpxczuwQyZLBUyygsA9Nsv95hpzUWwnNTVQmAq3xuh7vNwCp07UTmE5i11XAEExx4RA==}
|
||||
engines: {node: '>=18'}
|
||||
@@ -1950,6 +2048,20 @@ packages:
|
||||
|
||||
snapshots:
|
||||
|
||||
'@anthropic-ai/claude-agent-sdk@0.2.52(zod@3.25.76)':
|
||||
dependencies:
|
||||
zod: 3.25.76
|
||||
optionalDependencies:
|
||||
'@img/sharp-darwin-arm64': 0.34.5
|
||||
'@img/sharp-darwin-x64': 0.34.5
|
||||
'@img/sharp-linux-arm': 0.34.5
|
||||
'@img/sharp-linux-arm64': 0.34.5
|
||||
'@img/sharp-linux-x64': 0.34.5
|
||||
'@img/sharp-linuxmusl-arm64': 0.34.5
|
||||
'@img/sharp-linuxmusl-x64': 0.34.5
|
||||
'@img/sharp-win32-arm64': 0.34.5
|
||||
'@img/sharp-win32-x64': 0.34.5
|
||||
|
||||
'@appium/logger@1.7.1':
|
||||
dependencies:
|
||||
console-control-strings: 1.1.0
|
||||
@@ -2181,6 +2293,68 @@ snapshots:
|
||||
'@esbuild/win32-x64@0.27.2':
|
||||
optional: true
|
||||
|
||||
'@img/sharp-darwin-arm64@0.34.5':
|
||||
optionalDependencies:
|
||||
'@img/sharp-libvips-darwin-arm64': 1.2.4
|
||||
optional: true
|
||||
|
||||
'@img/sharp-darwin-x64@0.34.5':
|
||||
optionalDependencies:
|
||||
'@img/sharp-libvips-darwin-x64': 1.2.4
|
||||
optional: true
|
||||
|
||||
'@img/sharp-libvips-darwin-arm64@1.2.4':
|
||||
optional: true
|
||||
|
||||
'@img/sharp-libvips-darwin-x64@1.2.4':
|
||||
optional: true
|
||||
|
||||
'@img/sharp-libvips-linux-arm64@1.2.4':
|
||||
optional: true
|
||||
|
||||
'@img/sharp-libvips-linux-arm@1.2.4':
|
||||
optional: true
|
||||
|
||||
'@img/sharp-libvips-linux-x64@1.2.4':
|
||||
optional: true
|
||||
|
||||
'@img/sharp-libvips-linuxmusl-arm64@1.2.4':
|
||||
optional: true
|
||||
|
||||
'@img/sharp-libvips-linuxmusl-x64@1.2.4':
|
||||
optional: true
|
||||
|
||||
'@img/sharp-linux-arm64@0.34.5':
|
||||
optionalDependencies:
|
||||
'@img/sharp-libvips-linux-arm64': 1.2.4
|
||||
optional: true
|
||||
|
||||
'@img/sharp-linux-arm@0.34.5':
|
||||
optionalDependencies:
|
||||
'@img/sharp-libvips-linux-arm': 1.2.4
|
||||
optional: true
|
||||
|
||||
'@img/sharp-linux-x64@0.34.5':
|
||||
optionalDependencies:
|
||||
'@img/sharp-libvips-linux-x64': 1.2.4
|
||||
optional: true
|
||||
|
||||
'@img/sharp-linuxmusl-arm64@0.34.5':
|
||||
optionalDependencies:
|
||||
'@img/sharp-libvips-linuxmusl-arm64': 1.2.4
|
||||
optional: true
|
||||
|
||||
'@img/sharp-linuxmusl-x64@0.34.5':
|
||||
optionalDependencies:
|
||||
'@img/sharp-libvips-linuxmusl-x64': 1.2.4
|
||||
optional: true
|
||||
|
||||
'@img/sharp-win32-arm64@0.34.5':
|
||||
optional: true
|
||||
|
||||
'@img/sharp-win32-x64@0.34.5':
|
||||
optional: true
|
||||
|
||||
'@inquirer/external-editor@1.0.3(@types/node@20.19.28)':
|
||||
dependencies:
|
||||
chardet: 2.1.1
|
||||
|
||||
@@ -0,0 +1,216 @@
|
||||
---
|
||||
name: dogfood
|
||||
description: Systematically explore and test a web application to find bugs, UX issues, and other problems. Use when asked to "dogfood", "QA", "exploratory test", "find issues", "bug hunt", "test this app/site/platform", or review the quality of a web application. Produces a structured report with full reproduction evidence -- step-by-step screenshots, repro videos, and detailed repro steps for every issue -- so findings can be handed directly to the responsible teams.
|
||||
allowed-tools: Bash(agent-browser:*), Bash(npx agent-browser:*)
|
||||
---
|
||||
|
||||
# Dogfood
|
||||
|
||||
Systematically explore a web application, find issues, and produce a report with full reproduction evidence for every finding.
|
||||
|
||||
## Setup
|
||||
|
||||
Only the **Target URL** is required. Everything else has sensible defaults -- use them unless the user explicitly provides an override.
|
||||
|
||||
| Parameter | Default | Example override |
|
||||
|-----------|---------|-----------------|
|
||||
| **Target URL** | _(required)_ | `vercel.com`, `http://localhost:3000` |
|
||||
| **Session name** | Slugified domain (e.g., `vercel.com` -> `vercel-com`) | `--session my-session` |
|
||||
| **Output directory** | `./dogfood-output/` | `Output directory: /tmp/qa` |
|
||||
| **Scope** | Full app | `Focus on the billing page` |
|
||||
| **Authentication** | None | `Sign in to user@example.com` |
|
||||
|
||||
If the user says something like "dogfood vercel.com", start immediately with defaults. Do not ask clarifying questions unless authentication is mentioned but credentials are missing.
|
||||
|
||||
Always use `agent-browser` directly -- never `npx agent-browser`. The direct binary uses the fast Rust client. `npx` routes through Node.js and is significantly slower.
|
||||
|
||||
## Workflow
|
||||
|
||||
```
|
||||
1. Initialize Set up session, output dirs, report file
|
||||
2. Authenticate Sign in if needed, save state
|
||||
3. Orient Navigate to starting point, take initial snapshot
|
||||
4. Explore Systematically visit pages and test features
|
||||
5. Document Screenshot + record each issue as found
|
||||
6. Wrap up Update summary counts, close session
|
||||
```
|
||||
|
||||
### 1. Initialize
|
||||
|
||||
```bash
|
||||
mkdir -p {OUTPUT_DIR}/screenshots {OUTPUT_DIR}/videos
|
||||
```
|
||||
|
||||
Copy the report template into the output directory and fill in the header fields:
|
||||
|
||||
```bash
|
||||
cp {SKILL_DIR}/templates/dogfood-report-template.md {OUTPUT_DIR}/report.md
|
||||
```
|
||||
|
||||
Start a named session:
|
||||
|
||||
```bash
|
||||
agent-browser --session {SESSION} open {TARGET_URL}
|
||||
agent-browser --session {SESSION} wait --load networkidle
|
||||
```
|
||||
|
||||
### 2. Authenticate
|
||||
|
||||
If the app requires login:
|
||||
|
||||
```bash
|
||||
agent-browser --session {SESSION} snapshot -i
|
||||
# Identify login form refs, fill credentials
|
||||
agent-browser --session {SESSION} fill @e1 "{EMAIL}"
|
||||
agent-browser --session {SESSION} fill @e2 "{PASSWORD}"
|
||||
agent-browser --session {SESSION} click @e3
|
||||
agent-browser --session {SESSION} wait --load networkidle
|
||||
```
|
||||
|
||||
For OTP/email codes: ask the user, wait for their response, then enter the code.
|
||||
|
||||
After successful login, save state for potential reuse:
|
||||
|
||||
```bash
|
||||
agent-browser --session {SESSION} state save {OUTPUT_DIR}/auth-state.json
|
||||
```
|
||||
|
||||
### 3. Orient
|
||||
|
||||
Take an initial annotated screenshot and snapshot to understand the app structure:
|
||||
|
||||
```bash
|
||||
agent-browser --session {SESSION} screenshot --annotate {OUTPUT_DIR}/screenshots/initial.png
|
||||
agent-browser --session {SESSION} snapshot -i
|
||||
```
|
||||
|
||||
Identify the main navigation elements and map out the sections to visit.
|
||||
|
||||
### 4. Explore
|
||||
|
||||
Read [references/issue-taxonomy.md](references/issue-taxonomy.md) for the full list of what to look for and the exploration checklist.
|
||||
|
||||
**Strategy -- work through the app systematically:**
|
||||
|
||||
- Start from the main navigation. Visit each top-level section.
|
||||
- Within each section, test interactive elements: click buttons, fill forms, open dropdowns/modals.
|
||||
- Check edge cases: empty states, error handling, boundary inputs.
|
||||
- Try realistic end-to-end workflows (create, edit, delete flows).
|
||||
- Check the browser console for errors periodically.
|
||||
|
||||
**At each page:**
|
||||
|
||||
```bash
|
||||
agent-browser --session {SESSION} snapshot -i
|
||||
agent-browser --session {SESSION} screenshot --annotate {OUTPUT_DIR}/screenshots/{page-name}.png
|
||||
agent-browser --session {SESSION} errors
|
||||
agent-browser --session {SESSION} console
|
||||
```
|
||||
|
||||
Use your judgment on how deep to go. Spend more time on core features and less on peripheral pages. If you find a cluster of issues in one area, investigate deeper.
|
||||
|
||||
### 5. Document Issues (Repro-First)
|
||||
|
||||
Steps 4 and 5 happen together -- explore and document in a single pass. When you find an issue, stop exploring and document it immediately before moving on. Do not explore the whole app first and document later.
|
||||
|
||||
Every issue must be reproducible. When you find something wrong, do not just note it -- prove it with evidence. The goal is that someone reading the report can see exactly what happened and replay it.
|
||||
|
||||
**Choose the right level of evidence for the issue:**
|
||||
|
||||
#### Interactive / behavioral issues (functional, ux, console errors on action)
|
||||
|
||||
These require user interaction to reproduce -- use full repro with video and step-by-step screenshots:
|
||||
|
||||
1. **Start a repro video** _before_ reproducing:
|
||||
|
||||
```bash
|
||||
agent-browser --session {SESSION} record start {OUTPUT_DIR}/videos/issue-{NNN}-repro.webm
|
||||
```
|
||||
|
||||
2. **Walk through the steps at human pace.** Pause 1-2 seconds between actions so the video is watchable. Take a screenshot at each step:
|
||||
|
||||
```bash
|
||||
agent-browser --session {SESSION} screenshot {OUTPUT_DIR}/screenshots/issue-{NNN}-step-1.png
|
||||
sleep 1
|
||||
# Perform action (click, fill, etc.)
|
||||
sleep 1
|
||||
agent-browser --session {SESSION} screenshot {OUTPUT_DIR}/screenshots/issue-{NNN}-step-2.png
|
||||
sleep 1
|
||||
# ...continue until the issue manifests
|
||||
```
|
||||
|
||||
3. **Capture the broken state.** Pause so the viewer can see it, then take an annotated screenshot:
|
||||
|
||||
```bash
|
||||
sleep 2
|
||||
agent-browser --session {SESSION} screenshot --annotate {OUTPUT_DIR}/screenshots/issue-{NNN}-result.png
|
||||
```
|
||||
|
||||
4. **Stop the video:**
|
||||
|
||||
```bash
|
||||
agent-browser --session {SESSION} record stop
|
||||
```
|
||||
|
||||
5. Write numbered repro steps in the report, each referencing its screenshot.
|
||||
|
||||
#### Static / visible-on-load issues (typos, placeholder text, clipped text, misalignment, console errors on load)
|
||||
|
||||
These are visible without interaction -- a single annotated screenshot is sufficient. No video, no multi-step repro:
|
||||
|
||||
```bash
|
||||
agent-browser --session {SESSION} screenshot --annotate {OUTPUT_DIR}/screenshots/issue-{NNN}.png
|
||||
```
|
||||
|
||||
Write a brief description and reference the screenshot in the report. Set **Repro Video** to `N/A`.
|
||||
|
||||
---
|
||||
|
||||
**For all issues:**
|
||||
|
||||
1. **Append to the report immediately.** Do not batch issues for later. Write each one as you find it so nothing is lost if the session is interrupted.
|
||||
|
||||
2. **Increment the issue counter** (ISSUE-001, ISSUE-002, ...).
|
||||
|
||||
### 6. Wrap Up
|
||||
|
||||
Aim to find **5-10 well-documented issues**, then wrap up. Depth of evidence matters more than total count -- 5 issues with full repro beats 20 with vague descriptions.
|
||||
|
||||
After exploring:
|
||||
|
||||
1. Re-read the report and update the summary severity counts so they match the actual issues. Every `### ISSUE-` block must be reflected in the totals.
|
||||
2. Close the session:
|
||||
|
||||
```bash
|
||||
agent-browser --session {SESSION} close
|
||||
```
|
||||
|
||||
3. Tell the user the report is ready and summarize findings: total issues, breakdown by severity, and the most critical items.
|
||||
|
||||
## Guidance
|
||||
|
||||
- **Repro is everything.** Every issue needs proof -- but match the evidence to the issue. Interactive bugs need video and step-by-step screenshots. Static bugs (typos, placeholder text, visual glitches visible on load) only need a single annotated screenshot.
|
||||
- **Don't record video for static issues.** A typo or clipped text doesn't benefit from a video. Save video for issues that involve user interaction, timing, or state changes.
|
||||
- **For interactive issues, screenshot each step.** Capture the before, the action, and the after -- so someone can see the full sequence.
|
||||
- **Write repro steps that map to screenshots.** Each numbered step in the report should reference its corresponding screenshot. A reader should be able to follow the steps visually without touching a browser.
|
||||
- **Be thorough but use judgment.** You are not following a test script -- you are exploring like a real user would. If something feels off, investigate.
|
||||
- **Write findings incrementally.** Append each issue to the report as you discover it. If the session is interrupted, findings are preserved. Never batch all issues for the end.
|
||||
- **Never delete output files.** Do not `rm` screenshots, videos, or the report mid-session. Do not close the session and restart. Work forward, not backward.
|
||||
- **Never read the target app's source code.** You are testing as a user, not auditing code. Do not read HTML, JS, or config files of the app under test. All findings must come from what you observe in the browser.
|
||||
- **Check the console.** Many issues are invisible in the UI but show up as JS errors or failed requests.
|
||||
- **Test like a user, not a robot.** Try common workflows end-to-end. Click things a real user would click. Enter realistic data.
|
||||
- **Type like a human.** When filling form fields during video recording, use `type` instead of `fill` -- it types character-by-character. Use `fill` only outside of video recording when speed matters.
|
||||
- **Pace repro videos for humans.** Add `sleep 1` between actions and `sleep 2` before the final result screenshot. Videos should be watchable at 1x speed -- a human reviewing the report needs to see what happened, not a blur of instant state changes.
|
||||
- **Be efficient with commands.** Batch multiple `agent-browser` commands in a single shell call when they are independent (e.g., `agent-browser ... screenshot ... && agent-browser ... console`). Use `agent-browser --session {SESSION} scroll down 300` for scrolling -- do not use `key` or `evaluate` to scroll.
|
||||
|
||||
## References
|
||||
|
||||
| Reference | When to Read |
|
||||
|-----------|--------------|
|
||||
| [references/issue-taxonomy.md](references/issue-taxonomy.md) | Start of session -- calibrate what to look for, severity levels, exploration checklist |
|
||||
|
||||
## Templates
|
||||
|
||||
| Template | Purpose |
|
||||
|----------|---------|
|
||||
| [templates/dogfood-report-template.md](templates/dogfood-report-template.md) | Copy into output directory as the report file |
|
||||
@@ -0,0 +1,109 @@
|
||||
# Issue Taxonomy
|
||||
|
||||
Reference for categorizing issues found during dogfooding. Read this at the start of a dogfood session to calibrate what to look for.
|
||||
|
||||
## Contents
|
||||
|
||||
- [Severity Levels](#severity-levels)
|
||||
- [Categories](#categories)
|
||||
- [Exploration Checklist](#exploration-checklist)
|
||||
|
||||
## Severity Levels
|
||||
|
||||
| Severity | Definition |
|
||||
|----------|------------|
|
||||
| **critical** | Blocks a core workflow, causes data loss, or crashes the app |
|
||||
| **high** | Major feature broken or unusable, no workaround |
|
||||
| **medium** | Feature works but with noticeable problems, workaround exists |
|
||||
| **low** | Minor cosmetic or polish issue |
|
||||
|
||||
## Categories
|
||||
|
||||
### Visual / UI
|
||||
|
||||
- Layout broken or misaligned elements
|
||||
- Overlapping or clipped text
|
||||
- Inconsistent spacing, padding, or margins
|
||||
- Missing or broken icons/images
|
||||
- Dark mode / light mode rendering issues
|
||||
- Responsive layout problems (viewport sizes)
|
||||
- Z-index stacking issues (elements hidden behind others)
|
||||
- Font rendering issues (wrong font, size, weight)
|
||||
- Color contrast problems
|
||||
- Animation glitches or jank
|
||||
|
||||
### Functional
|
||||
|
||||
- Broken links (404, wrong destination)
|
||||
- Buttons or controls that do nothing on click
|
||||
- Form validation that rejects valid input or accepts invalid input
|
||||
- Incorrect redirects
|
||||
- Features that fail silently
|
||||
- State not persisted when expected (lost on refresh, navigation)
|
||||
- Race conditions (double-submit, stale data)
|
||||
- Broken search or filtering
|
||||
- Pagination issues
|
||||
- File upload/download failures
|
||||
|
||||
### UX
|
||||
|
||||
- Confusing or unclear navigation
|
||||
- Missing loading indicators or feedback after actions
|
||||
- Slow or unresponsive interactions (>300ms perceived delay)
|
||||
- Unclear error messages
|
||||
- Missing confirmation for destructive actions
|
||||
- Dead ends (no way to go back or proceed)
|
||||
- Inconsistent patterns across similar features
|
||||
- Missing keyboard shortcuts or focus management
|
||||
- Unintuitive defaults
|
||||
- Missing empty states or unhelpful empty states
|
||||
|
||||
### Content
|
||||
|
||||
- Typos or grammatical errors
|
||||
- Outdated or incorrect text
|
||||
- Placeholder or lorem ipsum content left in
|
||||
- Truncated text without tooltip or expansion
|
||||
- Missing or wrong labels
|
||||
- Inconsistent terminology
|
||||
|
||||
### Performance
|
||||
|
||||
- Slow page loads (>3s)
|
||||
- Janky scrolling or animations
|
||||
- Large layout shifts (content jumping)
|
||||
- Excessive network requests (check via console/network)
|
||||
- Memory leaks (page slows over time)
|
||||
- Unoptimized images (large file sizes)
|
||||
|
||||
### Console / Errors
|
||||
|
||||
- JavaScript exceptions in console
|
||||
- Failed network requests (4xx, 5xx)
|
||||
- Deprecation warnings
|
||||
- CORS errors
|
||||
- Mixed content warnings
|
||||
- Unhandled promise rejections
|
||||
|
||||
### Accessibility
|
||||
|
||||
- Missing alt text on images
|
||||
- Unlabeled form inputs
|
||||
- Poor keyboard navigation (can't tab to elements)
|
||||
- Focus traps
|
||||
- Insufficient color contrast
|
||||
- Missing ARIA attributes on dynamic content
|
||||
- Screen reader incompatible patterns
|
||||
|
||||
## Exploration Checklist
|
||||
|
||||
Use this as a guide for what to test on each page/feature:
|
||||
|
||||
1. **Visual scan** -- Take an annotated screenshot. Look for layout, alignment, and rendering issues.
|
||||
2. **Interactive elements** -- Click every button, link, and control. Do they work? Is there feedback?
|
||||
3. **Forms** -- Fill and submit. Test empty submission, invalid input, and edge cases.
|
||||
4. **Navigation** -- Follow all navigation paths. Check breadcrumbs, back button, deep links.
|
||||
5. **States** -- Check empty states, loading states, error states, and full/overflow states.
|
||||
6. **Console** -- Check for JS errors, failed requests, and warnings.
|
||||
7. **Responsiveness** -- If relevant, test at different viewport sizes.
|
||||
8. **Auth boundaries** -- Test what happens when not logged in, with different roles if applicable.
|
||||
@@ -0,0 +1,53 @@
|
||||
# Dogfood Report: {APP_NAME}
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| **Date** | {DATE} |
|
||||
| **App URL** | {URL} |
|
||||
| **Session** | {SESSION_NAME} |
|
||||
| **Scope** | {SCOPE} |
|
||||
|
||||
## Summary
|
||||
|
||||
| Severity | Count |
|
||||
|----------|-------|
|
||||
| Critical | 0 |
|
||||
| High | 0 |
|
||||
| Medium | 0 |
|
||||
| Low | 0 |
|
||||
| **Total** | **0** |
|
||||
|
||||
## Issues
|
||||
|
||||
<!-- Copy this block for each issue found. Interactive issues need video + step-by-step screenshots. Static issues (typos, visual glitches) only need a single screenshot -- set Repro Video to N/A. -->
|
||||
|
||||
### ISSUE-001: {Short title}
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| **Severity** | critical / high / medium / low |
|
||||
| **Category** | visual / functional / ux / content / performance / console / accessibility |
|
||||
| **URL** | {page URL where issue was found} |
|
||||
| **Repro Video** | {path to video, or N/A for static issues} |
|
||||
|
||||
**Description**
|
||||
|
||||
{What is wrong, what was expected, and what actually happened.}
|
||||
|
||||
**Repro Steps**
|
||||
|
||||
<!-- Each step has a screenshot. A reader should be able to follow along visually. -->
|
||||
|
||||
1. Navigate to {URL}
|
||||

|
||||
|
||||
2. {Action -- e.g., click "Settings" in the sidebar}
|
||||

|
||||
|
||||
3. {Action -- e.g., type "test" in the search field and press Enter}
|
||||

|
||||
|
||||
4. **Observe:** {what goes wrong -- e.g., the page shows a blank white screen instead of search results}
|
||||

|
||||
|
||||
---
|
||||
@@ -0,0 +1,261 @@
|
||||
import { describe, it, expect, beforeAll } from 'vitest';
|
||||
import { query } from '@anthropic-ai/claude-agent-sdk';
|
||||
import type { SDKMessage, SDKResultMessage } from '@anthropic-ai/claude-agent-sdk';
|
||||
import { mkdirSync, readFileSync, writeFileSync, appendFileSync, existsSync, readdirSync, rmSync } from 'node:fs';
|
||||
import path from 'node:path';
|
||||
|
||||
const AI_GATEWAY_URL =
|
||||
process.env.ANTHROPIC_BASE_URL || 'https://ai-gateway.vercel.sh';
|
||||
const API_KEY = process.env.AI_GATEWAY_API_KEY;
|
||||
const MODEL = process.env.DOGFOOD_MODEL || 'anthropic/claude-haiku-4.5';
|
||||
const CUSTOM_URL = process.env.DOGFOOD_URL;
|
||||
|
||||
const FIXTURE_PATH = path.resolve('test/e2e/fixtures/buggy-app.html');
|
||||
const SKILL_PATH = path.resolve('skills/dogfood/SKILL.md');
|
||||
const TARGET_URL = CUSTOM_URL || `file://${FIXTURE_PATH}`;
|
||||
const IS_FIXTURE = !CUSTOM_URL;
|
||||
|
||||
const OUTPUT_DIR = path.resolve('test/e2e/.dogfood-output');
|
||||
const EVAL_TIMEOUT = 10 * 60 * 1000;
|
||||
|
||||
async function runDogfood(outputDir: string): Promise<{
|
||||
result: SDKResultMessage | null;
|
||||
messages: SDKMessage[];
|
||||
toolsUsed: Set<string>;
|
||||
}> {
|
||||
const instruction = [
|
||||
`Read the dogfood skill at ${SKILL_PATH} and follow its workflow.`,
|
||||
`Dogfood ${TARGET_URL}`,
|
||||
`Output directory: ${outputDir}`,
|
||||
].join(' ');
|
||||
|
||||
const messages: SDKMessage[] = [];
|
||||
const toolsUsed = new Set<string>();
|
||||
let result: SDKResultMessage | null = null;
|
||||
|
||||
const conversation = query({
|
||||
prompt: instruction,
|
||||
options: {
|
||||
model: MODEL,
|
||||
cwd: process.cwd(),
|
||||
allowedTools: ['Bash', 'Read', 'Write', 'Edit', 'Glob', 'Grep'],
|
||||
permissionMode: 'bypassPermissions',
|
||||
allowDangerouslySkipPermissions: true,
|
||||
maxTurns: 80,
|
||||
maxBudgetUsd: 2,
|
||||
settingSources: ['project'],
|
||||
persistSession: false,
|
||||
env: {
|
||||
...process.env,
|
||||
ANTHROPIC_BASE_URL: AI_GATEWAY_URL,
|
||||
ANTHROPIC_API_KEY: API_KEY,
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
const verbose = process.env.DOGFOOD_VERBOSE !== '0';
|
||||
const log = verbose ? (msg: string) => process.stderr.write(` [dogfood] ${msg}\n`) : () => {};
|
||||
|
||||
const chatLogPath = path.join(outputDir, 'chat-log.jsonl');
|
||||
writeFileSync(chatLogPath, '');
|
||||
|
||||
function appendToLog(entry: Record<string, unknown>) {
|
||||
appendFileSync(chatLogPath, JSON.stringify(entry) + '\n');
|
||||
}
|
||||
|
||||
for await (const message of conversation) {
|
||||
messages.push(message);
|
||||
|
||||
if (message.type === 'system' && message.subtype === 'init') {
|
||||
log(`session started (model: ${message.model})`);
|
||||
appendToLog({ type: 'system', subtype: 'init', model: message.model });
|
||||
}
|
||||
|
||||
if (message.type === 'assistant' && message.message?.content) {
|
||||
const logParts: Record<string, unknown>[] = [];
|
||||
for (const block of message.message.content) {
|
||||
if ('type' in block && block.type === 'tool_use') {
|
||||
toolsUsed.add(block.name);
|
||||
const input = block.input as Record<string, unknown>;
|
||||
let preview: string;
|
||||
if (block.name === 'Bash') {
|
||||
const cmd = String(input.command ?? '');
|
||||
const firstLine = cmd.split('\n').find(l => l.trim() && !l.trim().startsWith('#')) ?? cmd.split('\n')[0];
|
||||
preview = firstLine.trim().slice(0, 200);
|
||||
} else if (block.name === 'Write') {
|
||||
preview = String(input.file_path ?? input.path ?? '');
|
||||
} else if (block.name === 'Read') {
|
||||
preview = String(input.file_path ?? input.path ?? '');
|
||||
} else if (block.name === 'Edit') {
|
||||
preview = String(input.file_path ?? input.path ?? '');
|
||||
} else {
|
||||
preview = JSON.stringify(input).slice(0, 120);
|
||||
}
|
||||
log(`${block.name}: ${preview}`);
|
||||
logParts.push({ tool: block.name, input: block.input });
|
||||
}
|
||||
if ('type' in block && block.type === 'text' && block.text) {
|
||||
const line = block.text.split('\n')[0].slice(0, 120);
|
||||
if (line.trim()) log(line);
|
||||
logParts.push({ text: block.text });
|
||||
}
|
||||
}
|
||||
appendToLog({ type: 'assistant', content: logParts });
|
||||
}
|
||||
|
||||
if (message.type === 'result') {
|
||||
result = message;
|
||||
const cost = `$${message.total_cost_usd.toFixed(4)}`;
|
||||
const usage = message.usage;
|
||||
const cacheRead = usage.cache_read_input_tokens ?? 0;
|
||||
const cacheCreate = usage.cache_creation_input_tokens ?? 0;
|
||||
const inputTokens = usage.input_tokens ?? 0;
|
||||
const cacheInfo = cacheRead > 0
|
||||
? ` | cache: ${cacheRead} read, ${cacheCreate} created, ${inputTokens} uncached`
|
||||
: '';
|
||||
if (message.subtype === 'success') {
|
||||
log(`done (${message.num_turns} turns, ${cost}${cacheInfo})`);
|
||||
} else {
|
||||
log(`stopped: ${message.subtype} (${message.num_turns} turns, ${cost}${cacheInfo})`);
|
||||
}
|
||||
appendToLog({ type: 'result', subtype: message.subtype, num_turns: message.num_turns, cost: message.total_cost_usd });
|
||||
}
|
||||
}
|
||||
|
||||
log(`chat log: ${chatLogPath}`);
|
||||
|
||||
return { result, messages, toolsUsed };
|
||||
}
|
||||
|
||||
function findFiles(dir: string, ext: string): string[] {
|
||||
if (!existsSync(dir)) return [];
|
||||
return readdirSync(dir, { recursive: true })
|
||||
.map(String)
|
||||
.filter((f) => f.endsWith(ext));
|
||||
}
|
||||
|
||||
describe.skipIf(!API_KEY)('Dogfood e2e eval (Agent SDK)', () => {
|
||||
const outputDir = OUTPUT_DIR;
|
||||
let evalResult: Awaited<ReturnType<typeof runDogfood>>;
|
||||
|
||||
beforeAll(async () => {
|
||||
if (existsSync(outputDir)) {
|
||||
rmSync(outputDir, { recursive: true, force: true });
|
||||
}
|
||||
mkdirSync(outputDir, { recursive: true });
|
||||
evalResult = await runDogfood(outputDir);
|
||||
}, EVAL_TIMEOUT);
|
||||
|
||||
it('completes without hard failure', () => {
|
||||
expect(evalResult.result, 'No result message received').toBeTruthy();
|
||||
const acceptable = ['success', 'error_max_turns', 'error_max_budget_usd'];
|
||||
expect(
|
||||
acceptable,
|
||||
`Agent failed unexpectedly: ${evalResult.result!.subtype}`
|
||||
).toContain(evalResult.result!.subtype);
|
||||
});
|
||||
|
||||
it('used agent-browser via Bash tool', () => {
|
||||
expect(
|
||||
evalResult.toolsUsed.has('Bash'),
|
||||
'Agent never used Bash (needed for agent-browser commands)'
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
it('produced a report file', () => {
|
||||
const reportPath = path.join(outputDir, 'report.md');
|
||||
expect(existsSync(reportPath), 'report.md not found in output dir').toBe(
|
||||
true
|
||||
);
|
||||
});
|
||||
|
||||
it('found a minimum number of issues', () => {
|
||||
const reportPath = path.join(outputDir, 'report.md');
|
||||
if (!existsSync(reportPath)) return;
|
||||
const report = readFileSync(reportPath, 'utf-8');
|
||||
|
||||
const issueBlocks = report.match(/###\s+ISSUE-\d+/g) || [];
|
||||
if (IS_FIXTURE) {
|
||||
expect(
|
||||
issueBlocks.length,
|
||||
`Expected >=2 issues from fixture, found ${issueBlocks.length}`
|
||||
).toBeGreaterThanOrEqual(2);
|
||||
} else {
|
||||
expect(issueBlocks.length).toBeGreaterThanOrEqual(1);
|
||||
}
|
||||
});
|
||||
|
||||
it('each issue has required fields and repro evidence', () => {
|
||||
const reportPath = path.join(outputDir, 'report.md');
|
||||
if (!existsSync(reportPath)) return;
|
||||
const report = readFileSync(reportPath, 'utf-8');
|
||||
|
||||
const issueSections = report.split(/(?=###\s+ISSUE-\d+)/).slice(1);
|
||||
for (const section of issueSections) {
|
||||
const issueId = section.match(/ISSUE-\d+/)?.[0] ?? 'unknown';
|
||||
|
||||
expect(section, `${issueId}: missing Severity`).toMatch(
|
||||
/\*\*Severity\*\*/i
|
||||
);
|
||||
|
||||
const sevMatch = section.match(
|
||||
/\*\*Severity\*\*\s*\|?\s*(critical|high|medium|low)/i
|
||||
);
|
||||
expect(sevMatch, `${issueId}: invalid severity value`).toBeTruthy();
|
||||
|
||||
expect(section, `${issueId}: missing Category`).toMatch(
|
||||
/\*\*Category\*\*/i
|
||||
);
|
||||
|
||||
expect(section, `${issueId}: missing URL`).toMatch(/\*\*URL\*\*/i);
|
||||
|
||||
expect(section, `${issueId}: missing Repro Video field`).toMatch(
|
||||
/\*\*Repro Video\*\*/i
|
||||
);
|
||||
|
||||
const hasScreenshot = /!\[.*?\]\(.*?\)/.test(section);
|
||||
const hasReproSteps = /\*\*Repro Steps\*\*/i.test(section);
|
||||
expect(
|
||||
hasScreenshot || hasReproSteps,
|
||||
`${issueId}: needs either screenshot refs or repro steps`
|
||||
).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it('has a summary table with non-zero total', () => {
|
||||
const reportPath = path.join(outputDir, 'report.md');
|
||||
if (!existsSync(reportPath)) return;
|
||||
const report = readFileSync(reportPath, 'utf-8');
|
||||
|
||||
expect(report, 'Missing Summary section').toContain('## Summary');
|
||||
const totalMatch = report.match(/\*\*Total\*\*\s*\|?\s*\*\*(\d+)\*\*/);
|
||||
expect(totalMatch, 'Summary Total not found').toBeTruthy();
|
||||
if (totalMatch) {
|
||||
const total = parseInt(totalMatch[1], 10);
|
||||
expect(total, 'Summary Total should be > 0').toBeGreaterThan(0);
|
||||
}
|
||||
});
|
||||
|
||||
it('produced screenshot files', () => {
|
||||
const screenshotsDir = path.join(outputDir, 'screenshots');
|
||||
const screenshots = findFiles(screenshotsDir, '.png');
|
||||
expect(
|
||||
screenshots.length,
|
||||
'No screenshot files found in output'
|
||||
).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it('produced video files for interactive issues', () => {
|
||||
const reportPath = path.join(outputDir, 'report.md');
|
||||
if (!existsSync(reportPath)) return;
|
||||
const report = readFileSync(reportPath, 'utf-8');
|
||||
const hasVideoRefs = /videos\/issue-\d+/.test(report);
|
||||
if (!hasVideoRefs) return;
|
||||
const videosDir = path.join(outputDir, 'videos');
|
||||
const videos = findFiles(videosDir, '.webm');
|
||||
expect(
|
||||
videos.length,
|
||||
'Report references videos but none were found'
|
||||
).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,210 @@
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import { readFileSync, existsSync } from 'node:fs';
|
||||
import path from 'node:path';
|
||||
|
||||
const SKILL_DIR = path.resolve('skills/dogfood');
|
||||
const SKILL_MD = path.join(SKILL_DIR, 'SKILL.md');
|
||||
const TAXONOMY_MD = path.join(SKILL_DIR, 'references', 'issue-taxonomy.md');
|
||||
const TEMPLATE_MD = path.join(SKILL_DIR, 'templates', 'dogfood-report-template.md');
|
||||
|
||||
function readSkillFile(filePath: string): string {
|
||||
return readFileSync(filePath, 'utf-8');
|
||||
}
|
||||
|
||||
function parseFrontmatter(content: string): Record<string, string> {
|
||||
const match = content.match(/^---\n([\s\S]*?)\n---/);
|
||||
if (!match) return {};
|
||||
const fields: Record<string, string> = {};
|
||||
for (const line of match[1].split('\n')) {
|
||||
const colonIdx = line.indexOf(':');
|
||||
if (colonIdx > 0) {
|
||||
fields[line.slice(0, colonIdx).trim()] = line.slice(colonIdx + 1).trim();
|
||||
}
|
||||
}
|
||||
return fields;
|
||||
}
|
||||
|
||||
describe('Dogfood skill: file structure', () => {
|
||||
it('SKILL.md exists', () => {
|
||||
expect(existsSync(SKILL_MD)).toBe(true);
|
||||
});
|
||||
|
||||
it('references/issue-taxonomy.md exists', () => {
|
||||
expect(existsSync(TAXONOMY_MD)).toBe(true);
|
||||
});
|
||||
|
||||
it('templates/dogfood-report-template.md exists', () => {
|
||||
expect(existsSync(TEMPLATE_MD)).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Dogfood skill: SKILL.md frontmatter', () => {
|
||||
const content = readSkillFile(SKILL_MD);
|
||||
const frontmatter = parseFrontmatter(content);
|
||||
|
||||
it('has name field', () => {
|
||||
expect(frontmatter.name).toBe('dogfood');
|
||||
});
|
||||
|
||||
it('has description field', () => {
|
||||
expect(frontmatter.description).toBeTruthy();
|
||||
expect(frontmatter.description!.length).toBeGreaterThan(50);
|
||||
});
|
||||
|
||||
it('has allowed-tools field', () => {
|
||||
expect(frontmatter['allowed-tools']).toBeTruthy();
|
||||
expect(frontmatter['allowed-tools']).toContain('agent-browser');
|
||||
});
|
||||
});
|
||||
|
||||
describe('Dogfood skill: SKILL.md body references', () => {
|
||||
const content = readSkillFile(SKILL_MD);
|
||||
|
||||
it('references issue-taxonomy.md', () => {
|
||||
expect(content).toContain('references/issue-taxonomy.md');
|
||||
});
|
||||
|
||||
it('references dogfood-report-template.md', () => {
|
||||
expect(content).toContain('templates/dogfood-report-template.md');
|
||||
});
|
||||
|
||||
it('referenced files exist on disk', () => {
|
||||
const refPattern = /\[.*?\]\((references\/.*?\.md|templates\/.*?\.md)\)/g;
|
||||
const refs = [...content.matchAll(refPattern)].map((m) => m[1]);
|
||||
expect(refs.length).toBeGreaterThan(0);
|
||||
for (const ref of refs) {
|
||||
const fullPath = path.join(SKILL_DIR, ref);
|
||||
expect(existsSync(fullPath), `Missing: ${ref}`).toBe(true);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('Dogfood skill: report template', () => {
|
||||
const template = readSkillFile(TEMPLATE_MD);
|
||||
|
||||
it('has ISSUE- prefix in issue blocks', () => {
|
||||
expect(template).toContain('ISSUE-');
|
||||
});
|
||||
|
||||
it('has Severity field', () => {
|
||||
expect(template).toContain('**Severity**');
|
||||
});
|
||||
|
||||
it('has Category field', () => {
|
||||
expect(template).toContain('**Category**');
|
||||
});
|
||||
|
||||
it('has URL field', () => {
|
||||
expect(template).toContain('**URL**');
|
||||
});
|
||||
|
||||
it('has Repro Video field', () => {
|
||||
expect(template).toContain('**Repro Video**');
|
||||
});
|
||||
|
||||
it('has Repro Steps section', () => {
|
||||
expect(template).toContain('**Repro Steps**');
|
||||
});
|
||||
|
||||
it('has screenshot image references in repro steps', () => {
|
||||
expect(template).toMatch(/!\[.*?\]\(screenshots\//);
|
||||
});
|
||||
|
||||
it('lists all valid severity values', () => {
|
||||
expect(template).toMatch(/critical\s*\/\s*high\s*\/\s*medium\s*\/\s*low/);
|
||||
});
|
||||
|
||||
it('lists all valid category values', () => {
|
||||
const categoryLine = template
|
||||
.split('\n')
|
||||
.find((l) => l.includes('**Category**'));
|
||||
expect(categoryLine).toBeTruthy();
|
||||
for (const cat of [
|
||||
'visual',
|
||||
'functional',
|
||||
'ux',
|
||||
'content',
|
||||
'performance',
|
||||
'console',
|
||||
'accessibility',
|
||||
]) {
|
||||
expect(categoryLine!.toLowerCase()).toContain(cat);
|
||||
}
|
||||
});
|
||||
|
||||
it('has Summary table with severity counts', () => {
|
||||
expect(template).toContain('## Summary');
|
||||
for (const sev of ['Critical', 'High', 'Medium', 'Low', 'Total']) {
|
||||
expect(template).toContain(sev);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('Dogfood skill: issue taxonomy', () => {
|
||||
const taxonomy = readSkillFile(TAXONOMY_MD);
|
||||
|
||||
it('has severity level definitions', () => {
|
||||
expect(taxonomy).toContain('## Severity Levels');
|
||||
for (const sev of ['critical', 'high', 'medium', 'low']) {
|
||||
expect(taxonomy.toLowerCase()).toContain(`**${sev}**`);
|
||||
}
|
||||
});
|
||||
|
||||
it('has all 7 category sections', () => {
|
||||
const expectedCategories = [
|
||||
'Visual',
|
||||
'Functional',
|
||||
'UX',
|
||||
'Content',
|
||||
'Performance',
|
||||
'Console',
|
||||
'Accessibility',
|
||||
];
|
||||
for (const cat of expectedCategories) {
|
||||
expect(taxonomy).toMatch(new RegExp(`###\\s+.*${cat}`, 'i'));
|
||||
}
|
||||
});
|
||||
|
||||
it('has exploration checklist', () => {
|
||||
expect(taxonomy).toContain('## Exploration Checklist');
|
||||
});
|
||||
|
||||
it('checklist has numbered items', () => {
|
||||
const checklistSection = taxonomy.split('## Exploration Checklist')[1];
|
||||
expect(checklistSection).toBeTruthy();
|
||||
const numberedItems = checklistSection!.match(/^\d+\./gm);
|
||||
expect(numberedItems!.length).toBeGreaterThanOrEqual(5);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Dogfood skill: cross-consistency', () => {
|
||||
const template = readSkillFile(TEMPLATE_MD);
|
||||
const taxonomy = readSkillFile(TAXONOMY_MD);
|
||||
|
||||
it('every category in template exists in taxonomy', () => {
|
||||
const categoryLine = template
|
||||
.split('\n')
|
||||
.find((l) => l.includes('**Category**'));
|
||||
expect(categoryLine).toBeTruthy();
|
||||
|
||||
const categories = categoryLine!
|
||||
.split('|')
|
||||
.pop()!
|
||||
.split('/')
|
||||
.map((c) => c.trim().toLowerCase())
|
||||
.filter(Boolean);
|
||||
|
||||
for (const cat of categories) {
|
||||
expect(
|
||||
taxonomy.toLowerCase(),
|
||||
`Category "${cat}" from template not found in taxonomy`
|
||||
).toMatch(new RegExp(`###\\s+.*${cat}`));
|
||||
}
|
||||
});
|
||||
|
||||
it('every severity in template exists in taxonomy', () => {
|
||||
for (const sev of ['critical', 'high', 'medium', 'low']) {
|
||||
expect(taxonomy.toLowerCase()).toContain(`**${sev}**`);
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,159 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>Buggy App - Dogfood Test Fixture</title>
|
||||
<style>
|
||||
* { margin: 0; padding: 0; box-sizing: border-box; }
|
||||
body { font-family: system-ui, sans-serif; color: #333; background: #f9f9f9; }
|
||||
header { background: #1a1a2e; color: #fff; padding: 16px 24px; display: flex; justify-content: space-between; align-items: center; }
|
||||
header h1 { font-size: 20px; }
|
||||
nav { display: flex; gap: 16px; }
|
||||
nav a { color: #ccc; text-decoration: none; }
|
||||
nav a:hover { color: #fff; }
|
||||
main { max-width: 960px; margin: 0 auto; padding: 32px 24px; }
|
||||
.card { background: #fff; border: 1px solid #e0e0e0; border-radius: 8px; padding: 24px; margin-bottom: 24px; }
|
||||
.card h2 { margin-bottom: 12px; }
|
||||
.btn { padding: 8px 16px; border: none; border-radius: 4px; cursor: pointer; font-size: 14px; }
|
||||
.btn-primary { background: #3b82f6; color: #fff; }
|
||||
.btn-danger { background: #ef4444; color: #fff; }
|
||||
input, textarea { padding: 8px 12px; border: 1px solid #d0d0d0; border-radius: 4px; font-size: 14px; width: 100%; margin-bottom: 12px; }
|
||||
label { display: block; margin-bottom: 4px; font-weight: 500; }
|
||||
footer { text-align: center; padding: 24px; color: #999; font-size: 12px; }
|
||||
|
||||
/* BUG: Visual - clipped text via overflow: hidden on a short container */
|
||||
.clipped-container {
|
||||
overflow: hidden;
|
||||
height: 20px;
|
||||
border: 1px solid #e0e0e0;
|
||||
padding: 4px 8px;
|
||||
margin-top: 8px;
|
||||
}
|
||||
|
||||
/* BUG: Visual - misaligned element */
|
||||
.misaligned {
|
||||
display: flex;
|
||||
align-items: flex-start; /* should be center */
|
||||
gap: 12px;
|
||||
padding: 12px;
|
||||
background: #f0f4ff;
|
||||
border-radius: 8px;
|
||||
margin-top: 12px;
|
||||
}
|
||||
.misaligned .icon {
|
||||
width: 40px;
|
||||
height: 40px;
|
||||
background: #3b82f6;
|
||||
border-radius: 50%;
|
||||
flex-shrink: 0;
|
||||
margin-top: 14px; /* intentionally off */
|
||||
}
|
||||
.misaligned .label {
|
||||
font-size: 16px;
|
||||
line-height: 40px;
|
||||
}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
|
||||
<header>
|
||||
<h1>Buggy App</h1>
|
||||
<nav>
|
||||
<a href="#dashboard">Dashboard</a>
|
||||
<a href="#settings">Settings</a>
|
||||
<!-- BUG: Functional - broken link to nonexistent page -->
|
||||
<a href="#/this-page-does-not-exist">Reports</a>
|
||||
<a href="#help">Help</a>
|
||||
</nav>
|
||||
</header>
|
||||
|
||||
<main>
|
||||
<!-- BUG: Content - typo "Welocme" -->
|
||||
<h2 style="margin-bottom: 24px;">Welocme to the Dashboard</h2>
|
||||
|
||||
<!-- Card 1: Functional bug - button throws JS error -->
|
||||
<div class="card">
|
||||
<h2>Quick Actions</h2>
|
||||
<p>Perform common tasks from here.</p>
|
||||
<div style="margin-top: 12px; display: flex; gap: 8px;">
|
||||
<!-- BUG: Functional - button throws JS error on click -->
|
||||
<button class="btn btn-primary" onclick="processAction()">Run Analysis</button>
|
||||
<button class="btn btn-danger" onclick="deleteAllData()">Delete All Data</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Card 2: Visual bugs - clipped text and misaligned element -->
|
||||
<div class="card">
|
||||
<h2>System Status</h2>
|
||||
<!-- BUG: Visual - text is clipped because container is too short -->
|
||||
<div class="clipped-container">
|
||||
The system is currently operating normally. All services are online and responding within expected latency thresholds. Last health check completed at 14:32 UTC.
|
||||
</div>
|
||||
<!-- BUG: Visual - icon and label are misaligned -->
|
||||
<div class="misaligned">
|
||||
<div class="icon"></div>
|
||||
<span class="label">All systems operational</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Card 3: Content bug - placeholder text -->
|
||||
<div class="card">
|
||||
<h2>Recent Activity</h2>
|
||||
<!-- BUG: Content - lorem ipsum placeholder left in -->
|
||||
<p>Lorem ipsum dolor sit amet, consectetur adipiscing elit. Sed do eiusmod tempor incididunt ut labore et dolore magna aliqua. Ut enim ad minim veniam, quis nostrud exercitation ullamco laboris.</p>
|
||||
</div>
|
||||
|
||||
<!-- Card 4: UX bug - form with no feedback on submit -->
|
||||
<div class="card">
|
||||
<h2>Contact Support</h2>
|
||||
<form id="support-form">
|
||||
<label for="subject">Subject</label>
|
||||
<input type="text" id="subject" placeholder="Enter subject">
|
||||
<label for="message">Message</label>
|
||||
<textarea id="message" rows="3" placeholder="Describe your issue"></textarea>
|
||||
<!-- BUG: UX - submit does nothing, no feedback -->
|
||||
<button type="submit" class="btn btn-primary">Send Message</button>
|
||||
</form>
|
||||
</div>
|
||||
|
||||
<!-- Card 5: UX bug - empty state with no message -->
|
||||
<div class="card">
|
||||
<h2>Notifications</h2>
|
||||
<!-- BUG: UX - empty container with no empty state message -->
|
||||
<div id="notifications-list" style="min-height: 60px;">
|
||||
</div>
|
||||
</div>
|
||||
</main>
|
||||
|
||||
<footer>
|
||||
© 2025 Buggy App Inc. All rights reserved.
|
||||
</footer>
|
||||
|
||||
<script>
|
||||
// BUG: Console - error on page load
|
||||
console.error("Failed to initialize analytics: endpoint not configured");
|
||||
|
||||
// BUG: Console - failed fetch on page load
|
||||
fetch("https://api.nonexistent-endpoint.invalid/v1/health")
|
||||
.catch(function() {});
|
||||
|
||||
// BUG: Functional - function referenced by button is broken
|
||||
function processAction() {
|
||||
// Throws because undefinedService is not defined
|
||||
undefinedService.runAnalysis();
|
||||
}
|
||||
|
||||
// No confirmation for destructive action
|
||||
function deleteAllData() {
|
||||
alert("All data deleted!");
|
||||
}
|
||||
|
||||
// Form submit does nothing
|
||||
document.getElementById("support-form").addEventListener("submit", function(e) {
|
||||
e.preventDefault();
|
||||
});
|
||||
</script>
|
||||
|
||||
</body>
|
||||
</html>
|
||||
+1
-1
@@ -3,7 +3,7 @@ import { defineConfig } from 'vitest/config';
|
||||
export default defineConfig({
|
||||
test: {
|
||||
globals: true,
|
||||
include: ['src/**/*.test.ts', 'test/**/*.test.ts'],
|
||||
include: ['src/**/*.test.ts', 'test/**/*.test.ts', 'test/**/*.eval.ts'],
|
||||
testTimeout: 30000,
|
||||
},
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user