{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-haiku-4-5-20251001",
    "model_release_date": "2025-10-01",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T21:27:24.200Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-4-5-20251001",
      "reported_models": [
        "claude-haiku-4-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T21:27:24.200Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-4-5-20251001": {
          "input_per_mtok": 1,
          "output_per_mtok": 5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.827778,
        "mean": 0.827778,
        "stddev": 0.003849,
        "samples": [
          0.82,
          0.82,
          0.83
        ],
        "generation_hash": "5e2efa427da2b27d341990d82b94066eee5cc9d331adec4e756e26a6aec594f2",
        "judge_sample_hashes": [
          "fce1416aa651283c1273c28593935ab54aee15ff2212117f83db9fbe0f4d0c3e",
          "277d4050aa9f3a91894b0562d46d990ba9fe7a1348e8f8141d97ffcc0e437ce1",
          "5b02f308e99e0ad5256e492e9119a9afa7a202a7315542d9bcd4d5d044900d60"
        ],
        "threshold": 0.7,
        "reason": "Uses conventional `fix:` prefix with a specific subject, and the body explains the user-facing symptom (1h vs 24h expiry) and the unit-conversion cause; slightly redundant restatement keeps it just above baseline.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43785,
          "output_tokens": 1274,
          "cached_tokens": 32445,
          "wall_ms": 25874
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.827778,
          "sd": 0.003849,
          "judge_sd_mean": 0.008591,
          "variance_ratio": 0.448027,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "3045cf802113180dbc833cdea4fa3260570ed3eb5bc40e4e2f4ad23131722ce6",
              "status": "measured",
              "samples": [
                0.84,
                0.82,
                0.83
              ],
              "judge_sample_hashes": [
                "25d43d1f579f8b68ae8e1bccb44c3de20a620c303fca9d48e7d3825819aa499c",
                "df7ffc87f4dfe72bea019f55ec7cb4e2113779f267df36987051c629470f4730",
                "7aec08bd2424ede71d743e51b309b533294045f51581852d5a54523b2cba61d5"
              ],
              "mean": 0.83,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 17698,
                "output_tokens": 2022,
                "cached_tokens": 17688,
                "wall_ms": 20079
              },
              "judge_usage": {
                "input_tokens": 43695,
                "output_tokens": 1010,
                "cached_tokens": 32385,
                "wall_ms": 55513
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "7cc7c629023531c518ba15dbc92ac5ad2c1a9fe2d08f3b36373d6f39c12e6be3",
              "status": "measured",
              "samples": [
                0.83,
                0.82,
                0.84
              ],
              "judge_sample_hashes": [
                "c52ef32aafecebdb26abb62616ea15cc46f8cd87afae8b5909829cbe9666e4e6",
                "bf479e27aee62fabdd79bb463f860259a57df821d7a88560eb0f8a2867bd9cb8",
                "32b7c0e00a43f6948f642be511a347b12f7b9c5b195ba4efc521eb6c7471ad56"
              ],
              "mean": 0.83,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 17698,
                "output_tokens": 1323,
                "cached_tokens": 17688,
                "wall_ms": 14558
              },
              "judge_usage": {
                "input_tokens": 43617,
                "output_tokens": 1210,
                "cached_tokens": 32333,
                "wall_ms": 26183
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "5e2efa427da2b27d341990d82b94066eee5cc9d331adec4e756e26a6aec594f2",
              "status": "measured",
              "samples": [
                0.82,
                0.82,
                0.83
              ],
              "judge_sample_hashes": [
                "fce1416aa651283c1273c28593935ab54aee15ff2212117f83db9fbe0f4d0c3e",
                "277d4050aa9f3a91894b0562d46d990ba9fe7a1348e8f8141d97ffcc0e437ce1",
                "5b02f308e99e0ad5256e492e9119a9afa7a202a7315542d9bcd4d5d044900d60"
              ],
              "mean": 0.823333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 17698,
                "output_tokens": 2651,
                "cached_tokens": 17688,
                "wall_ms": 26120
              },
              "judge_usage": {
                "input_tokens": 43785,
                "output_tokens": 1274,
                "cached_tokens": 32445,
                "wall_ms": 25874
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.837778,
        "mean": 0.837778,
        "stddev": 0.015396,
        "samples": [
          0.85,
          0.85,
          0.84
        ],
        "generation_hash": "89e4bf4f8e7f7b67bd015b7cfbbc613865287bf1a1954994ea1265f7d9b5fb8b",
        "judge_sample_hashes": [
          "4f292817947136b05aa6b160e5c52d5c7312e12338e0db72f71f2263f134a3a2",
          "5a3e7714b238af0e46c2fdce09fa5793ada208368348aaa799ca5b5f2903b50e",
          "823629de0b2841ad630ecae19d1bcd02e2b99386f58dd6e0d50780c5a2459495"
        ],
        "threshold": 0.7,
        "reason": "Conventional `fix(auth):` prefix, specific imperative subject, and a body giving both the user-facing symptom (~1h vs 24h expiry) and the unit-conversion cause; slight redundancy in closing sentence.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43650,
          "output_tokens": 1031,
          "cached_tokens": 32355,
          "wall_ms": 22522
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.837778,
          "sd": 0.015396,
          "judge_sd_mean": 0.009623,
          "variance_ratio": 1.599917,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "869e280f60c4e9f82c7a78bdf3bd10a70049ed28a5dcb792756e7165ccda9436",
              "status": "measured",
              "samples": [
                0.8,
                0.83,
                0.83
              ],
              "judge_sample_hashes": [
                "9aa440328084cadb19065b0f49d1f5a1b5bb32f3a5331b5942001d03a8df7258",
                "cf035796de754aafc86bce36322865ff096e34a8c473980d6d518422569e66dd",
                "e61c27582a76fe13c444452faff65a125c5763678727faa8738d9ef4bea5a1a9"
              ],
              "mean": 0.82,
              "stddev": 0.017321,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14085,
                "output_tokens": 883,
                "cached_tokens": 14075,
                "wall_ms": 10790
              },
              "judge_usage": {
                "input_tokens": 43566,
                "output_tokens": 722,
                "cached_tokens": 32299,
                "wall_ms": 29820
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "7b3b2c560c2fdaf3fdc96e04a7c2e604e8d6c01571042a7a2c312f380682df4d",
              "status": "measured",
              "samples": [
                0.85,
                0.84,
                0.85
              ],
              "judge_sample_hashes": [
                "9dc32f6d305ef05f6dfdb1c95d10a81cdbaff75ea27ad551d41db2db9e1c49c7",
                "cd9b0832245b5622db690b5c4efbecbb822ac6e65ab4882469c8b16b16394b8f",
                "c84a67c9bd8e82518b70e67d025d6bab5e3cfbfaaed5e10eeb09e473971b9651"
              ],
              "mean": 0.846667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14085,
                "output_tokens": 630,
                "cached_tokens": 14075,
                "wall_ms": 8242
              },
              "judge_usage": {
                "input_tokens": 43632,
                "output_tokens": 1037,
                "cached_tokens": 32343,
                "wall_ms": 21472
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "89e4bf4f8e7f7b67bd015b7cfbbc613865287bf1a1954994ea1265f7d9b5fb8b",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.84
              ],
              "judge_sample_hashes": [
                "4f292817947136b05aa6b160e5c52d5c7312e12338e0db72f71f2263f134a3a2",
                "5a3e7714b238af0e46c2fdce09fa5793ada208368348aaa799ca5b5f2903b50e",
                "823629de0b2841ad630ecae19d1bcd02e2b99386f58dd6e0d50780c5a2459495"
              ],
              "mean": 0.846667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14085,
                "output_tokens": 572,
                "cached_tokens": 14075,
                "wall_ms": 8298
              },
              "judge_usage": {
                "input_tokens": 43650,
                "output_tokens": 1031,
                "cached_tokens": 32355,
                "wall_ms": 22522
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.827778,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.837778,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.827778,
    "baseline_score": 0.837778,
    "delta": -0.01,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "ce5e71feb67ea891c3af875da1d825f0699d6ac6fb6960973bbe4105d7ec92b0",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 17698,
      "mean_output_tokens": 1998.67,
      "mean_cost_usd_per_call": 0.027691,
      "median_wall_ms": 20079,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 14085,
      "mean_output_tokens": 695,
      "mean_cost_usd_per_call": 0.01756,
      "median_wall_ms": 8298,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.010131,
    "skill_incremental_cost_usd_per_1k_calls": 10.131,
    "output_tokens_delta": 1303.67,
    "median_wall_ms_delta": 11781,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.4948,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}