{
  "slug": "eval-harness",
  "title": "Eval Harness",
  "description": "Evaluation-driven development: define, run, and grade agent evals over time. Use when: \"eval harness\", \"evaluation framework\", \"test harness\", \"create evals\", \"benchmark accuracy\", \"run evals\", \"test agent quality\", \"regression check\", \"eval suite\", \"grade outputs\", \"eval report\", \"confidence interval on a pass rate\", \"how many test cases do I need\", \"is this improvement real\", \"is 90% good\", \"quantify the eval\", \"what pass rate is good enough\". PROACTIVE when building agent or LLM features needing quality gates.",
  "collection": "Made by Joy",
  "pack": null,
  "tags": {
    "tool": [
      "Claude Code",
      "Codex"
    ],
    "task": [
      "Workflow"
    ]
  },
  "author": "Joy Dong",
  "license": "MIT",
  "install": "npx skills add https://github.com/joydai2026-del/skills/tree/0d15d89c093f8f3852fe51befb3c364e387244f1/eval-harness",
  "source": "https://github.com/joydai2026-del/skills/tree/0d15d89c093f8f3852fe51befb3c364e387244f1/eval-harness",
  "page": "https://www.joydong.org/skills/eval-harness",
  "public_tree_sha256": "b7386ed3cfc6c3cd3fcc40981d53109f462a0b114f1ff97f220f3f4e7de3cd32",
  "files": [
    {
      "path": "SKILL.md",
      "bytes": 16238,
      "sha256": "d403172064761cb9c75a70d2bc1d073d1295ee20d8db9e9e612fb6dbda2e4156"
    }
  ],
  "teaching_published": false,
  "verified": {
    "status": "pass_install",
    "recorded_status": "pass",
    "last_run_at": "2026-09-11T01:24:01Z",
    "runner": "github-actions",
    "runner_digest": "sha256:b573c7946174ef251554e541b13e12f280b1c9977e3f17607042aa6715e90d36",
    "agent_versions": {
      "claude_code": "n/a",
      "codex": "n/a",
      "skills_cli": "1.5.25"
    }
  },
  "last_updated": "2026-09-10T18:56:00Z"
}
