{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-haiku-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T20:48:24.950Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-5-5",
      "reported_models": [
        "claude-haiku-5-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T20:48:24.950Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-5-5": {
          "input_per_mtok": 0.1,
          "output_per_mtok": 0.5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.838889,
        "mean": 0.838889,
        "stddev": 0.013878,
        "samples": [
          0.83,
          0.85,
          0.85
        ],
        "generation_hash": "c63277315850785d7adabd513cd14cc32fc3d0fb8f71da68af51b5b30cdfa6de",
        "judge_sample_hashes": [
          "9d2cda2696f90af987a01b8236391236e42d23a3b9a223903cd80fb8775c99a5",
          "6c8ba957ba79ea10f883aca664c140a0a1930d4f06293684fc48d83c04fd862b",
          "89c85a9fd7bcf778df248b3507e5dbdb2484f2e01b27225bc87c6e18d2108491"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix:` prefix with a specific subject, and the body states the user-facing consequence (links expiring in ~1h not 24h) and the wrong-unit cause.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43614,
          "output_tokens": 972,
          "cached_tokens": 32331,
          "wall_ms": 20297
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.838889,
          "sd": 0.013878,
          "judge_sd_mean": 0.010788,
          "variance_ratio": 1.286429,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "620f11c74bac7683aca32b524ce04b9a692aa5e8fe0fb5af3e0b3d8830b0ecb3",
              "status": "measured",
              "samples": [
                0.8,
                0.84,
                0.83
              ],
              "judge_sample_hashes": [
                "7c68e8fb59db38738c9ce5305a650a328eb36a88b7a62128ba071cbc91bd1697",
                "73de34cceea69453fa48f5180351702198ebad55bb15b81cbc95b7edc0fc4c3a",
                "27dcefd1cad609b16d11dad692f2b24443037cfcb15ff5127c5bbddfd4ff35a8"
              ],
              "mean": 0.823333,
              "stddev": 0.020817,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 24834,
                "output_tokens": 99,
                "cached_tokens": 24832,
                "wall_ms": 3481
              },
              "judge_usage": {
                "input_tokens": 43641,
                "output_tokens": 1833,
                "cached_tokens": 32349,
                "wall_ms": 32589
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "7f7c36fdad742e269392e50f094537fe377493ebe298882579fdf9a2a37514df",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "e2a50625b694ef3a8a2b822066c24ecc7c2622162074334b369fe985b708ba99",
                "708bc7b116f11e40105c47fb548bcecf36aaf2314af8d882f34b2bec63868d4b",
                "7d5a3616fb49b0966aa906ecd820f8af33081ec855be9dd143113c3f74d5faf7"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 24834,
                "output_tokens": 551,
                "cached_tokens": 24832,
                "wall_ms": 4859
              },
              "judge_usage": {
                "input_tokens": 43617,
                "output_tokens": 806,
                "cached_tokens": 32333,
                "wall_ms": 17638
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "c63277315850785d7adabd513cd14cc32fc3d0fb8f71da68af51b5b30cdfa6de",
              "status": "measured",
              "samples": [
                0.83,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "9d2cda2696f90af987a01b8236391236e42d23a3b9a223903cd80fb8775c99a5",
                "6c8ba957ba79ea10f883aca664c140a0a1930d4f06293684fc48d83c04fd862b",
                "89c85a9fd7bcf778df248b3507e5dbdb2484f2e01b27225bc87c6e18d2108491"
              ],
              "mean": 0.843333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 24834,
                "output_tokens": 645,
                "cached_tokens": 24832,
                "wall_ms": 5569
              },
              "judge_usage": {
                "input_tokens": 43614,
                "output_tokens": 972,
                "cached_tokens": 32331,
                "wall_ms": 20297
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.848889,
        "mean": 0.848889,
        "stddev": 0.011706,
        "samples": [
          0.85,
          0.82,
          0.84
        ],
        "generation_hash": "329245b5a3d613cb4b43c928f9b4e2d9b563cd0484fd32a65db8366fea04e11c",
        "judge_sample_hashes": [
          "dad57a270c260aae87559ea161114f19c499b076d4ab2a459d6bd9861e601f22",
          "1b5ce92fa4a9e2bf42815afe7e399cf6a056bdff83655922308a472b34acce8a",
          "f57fc0faab8313bedf2197c1f39a8df31d7326c27960f8c25dc261196b2bb34c"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix(auth):` prefix, specific subject, and body conveying both user-facing impact (1h vs 24h expiry) and cause (wrong time unit) actionably.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43803,
          "output_tokens": 1006,
          "cached_tokens": 32457,
          "wall_ms": 23476
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.848889,
          "sd": 0.011706,
          "judge_sd_mean": 0.008425,
          "variance_ratio": 1.389436,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "b58c89ff666a3ce989f334ec7c1d6de54240ea55d38ecdf85287ec313a8ee854",
              "status": "measured",
              "samples": [
                0.85,
                0.87,
                0.86
              ],
              "judge_sample_hashes": [
                "9e7df82ad8f1d7aa6460877c6e4894ef23bd571e7d1a127ed8a2767f9f4dce51",
                "af4f32b95b5f24026b1ff7fae1ccdad0abb9e57b2f67dfb671c3bb24356c8685",
                "c67aafbc440a90911491d4696be0b6569e779b8421ee49d01cbca71298fa56ff"
              ],
              "mean": 0.86,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19830,
                "output_tokens": 1044,
                "cached_tokens": 19828,
                "wall_ms": 6992
              },
              "judge_usage": {
                "input_tokens": 43872,
                "output_tokens": 1187,
                "cached_tokens": 32503,
                "wall_ms": 60900
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "571e92494ab3113f45408daa2f6e2efd5f096421c3c62ff6dce85f1b8c2b3368",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "922e3a9cb327f08e3bfaa1e5da3e6444f281641a9156c0353fba110b745d479a",
                "6fdd811bafd234cdbb809578ffd74ef6c29d3567139c732d7de7181b66f8e714",
                "8d6bbbae5104bd53633f7717ff52a4702f4166ab3e2dd800fdaecc2f22182ca5"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19830,
                "output_tokens": 619,
                "cached_tokens": 19828,
                "wall_ms": 5206
              },
              "judge_usage": {
                "input_tokens": 43635,
                "output_tokens": 832,
                "cached_tokens": 32345,
                "wall_ms": 18768
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "329245b5a3d613cb4b43c928f9b4e2d9b563cd0484fd32a65db8366fea04e11c",
              "status": "measured",
              "samples": [
                0.85,
                0.82,
                0.84
              ],
              "judge_sample_hashes": [
                "dad57a270c260aae87559ea161114f19c499b076d4ab2a459d6bd9861e601f22",
                "1b5ce92fa4a9e2bf42815afe7e399cf6a056bdff83655922308a472b34acce8a",
                "f57fc0faab8313bedf2197c1f39a8df31d7326c27960f8c25dc261196b2bb34c"
              ],
              "mean": 0.836667,
              "stddev": 0.015275,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19830,
                "output_tokens": 730,
                "cached_tokens": 19828,
                "wall_ms": 5974
              },
              "judge_usage": {
                "input_tokens": 43803,
                "output_tokens": 1006,
                "cached_tokens": 32457,
                "wall_ms": 23476
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.838889,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.848889,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.838889,
    "baseline_score": 0.848889,
    "delta": -0.01,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "5d46e61c93a1bc1d370cd96ab75775947213cd8ec8dae95a55909deebf4f4167",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 24834,
      "mean_output_tokens": 431.67,
      "mean_cost_usd_per_call": 0.002699,
      "median_wall_ms": 4859,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 19830,
      "mean_output_tokens": 797.67,
      "mean_cost_usd_per_call": 0.002382,
      "median_wall_ms": 5974,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.000317,
    "skill_incremental_cost_usd_per_1k_calls": 0.317,
    "output_tokens_delta": -366,
    "median_wall_ms_delta": -1115,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.486535,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}