{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-opus-5",
    "model_release_date": "2026-07-24",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-15T16:06:45.501Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5",
      "reported_models": [
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-15T16:06:45.501Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.862222,
        "mean": 0.862222,
        "stddev": 0.006939,
        "samples": [
          0.85,
          0.86,
          0.86
        ],
        "generation_hash": "726f4e70cf6cd1266314455a506a8fae7b0e39788b92eb3768bf1327cc63cba4",
        "judge_sample_hashes": [
          "4353fa557af9d76de22294edbd8ea33ada6f639a959f469e2dbf8b4a569cd104",
          "c6946eb6cfc58fac4682f1237cf4c82ef3249e3109659460cc825a6f890e5485",
          "f5d61b3336dbb881d43006f2e89e9368c8511a9ce5ba25286e103aeb6cef8cc8"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix:` with a specific subject; the body explains the cause (unconverted unit) and user impact (expired-link errors), so it is exemplary, though it never names the actual units.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 10227,
          "output_tokens": 600,
          "cached_tokens": 8973,
          "wall_ms": 16430
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.862222,
          "sd": 0.006939,
          "judge_sd_mean": 0.001925,
          "variance_ratio": 3.604675,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "7f2dc67862cf79fae915c8b65871ffb7b97eaed33a617b4da2211a09891d98e5",
              "status": "measured",
              "samples": [
                0.87,
                0.87,
                0.87
              ],
              "judge_sample_hashes": [
                "427001c9f94f175846e2025ecc581b50b02e2060a2d028c16825744c8c29ce14",
                "920a213fdb8da2d610dd17109234e5b8dd116728f782542148eba96ef8a658a7",
                "c097ec18cff17367bf84d69c1d5b913c91687b3e9a3a30d54b2a8737d521d779"
              ],
              "mean": 0.87,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 7557,
                "output_tokens": 398,
                "cached_tokens": 0,
                "wall_ms": 9001
              },
              "judge_usage": {
                "input_tokens": 10554,
                "output_tokens": 577,
                "cached_tokens": 9191,
                "wall_ms": 15701
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "a5a12b9e660b7dbc9f3bf32967255e9fba1edf32903194950471400551c4c6d3",
              "status": "measured",
              "samples": [
                0.86,
                0.86,
                0.86
              ],
              "judge_sample_hashes": [
                "81b5040840e19bac99b1a7c4b0a553414580f40ee192c41d31c1d218518421eb",
                "c42df4b207f7fbbd1d734df9d83baaa5035569109a15572f82068dd50dcdb0ff",
                "e7f7e5632f4af55f900dbc2d0fee9847f99e5258b1b812ebec7d77f5b5fb5cb9"
              ],
              "mean": 0.86,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 7557,
                "output_tokens": 455,
                "cached_tokens": 7555,
                "wall_ms": 9662
              },
              "judge_usage": {
                "input_tokens": 10494,
                "output_tokens": 282,
                "cached_tokens": 9151,
                "wall_ms": 12479
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "726f4e70cf6cd1266314455a506a8fae7b0e39788b92eb3768bf1327cc63cba4",
              "status": "measured",
              "samples": [
                0.85,
                0.86,
                0.86
              ],
              "judge_sample_hashes": [
                "4353fa557af9d76de22294edbd8ea33ada6f639a959f469e2dbf8b4a569cd104",
                "c6946eb6cfc58fac4682f1237cf4c82ef3249e3109659460cc825a6f890e5485",
                "f5d61b3336dbb881d43006f2e89e9368c8511a9ce5ba25286e103aeb6cef8cc8"
              ],
              "mean": 0.856667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 7557,
                "output_tokens": 330,
                "cached_tokens": 7555,
                "wall_ms": 7644
              },
              "judge_usage": {
                "input_tokens": 10227,
                "output_tokens": 600,
                "cached_tokens": 8973,
                "wall_ms": 16430
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.3,
        "mean": 0.3,
        "stddev": 0,
        "samples": [
          0.3,
          0.3,
          0.3
        ],
        "generation_hash": "172e537d119953f63177b9d7492edc5c1ff56b8238068a01917e43a6b86a2e92",
        "judge_sample_hashes": [
          "c3d73282997330f890f6353de483ce0e237c3cfdf68c009d2fb169132cb99799",
          "caaad1ee415feb18bb21f23c3a5e670e4a5327be04963f4f523e3a65ff5544c7",
          "7794a506e56c6dac5686b69669e5f5e1319a739ce0e95577fe1e9adfabe5b751"
        ],
        "threshold": 0.7,
        "reason": "Subject lacks the conventional `fix:` prefix, triggering the 0.3 cap, although the subject is specific and the body explains cause and user impact well.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 10188,
          "output_tokens": 274,
          "cached_tokens": 8947,
          "wall_ms": 12425
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.3,
          "sd": 0,
          "judge_sd_mean": 0,
          "variance_ratio": null,
          "variance_ratio_unavailable": "judge_sd_zero",
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "a1cd469d4279d8e31af660688f4dbc7c267606c9f8f66ff470a094e0345f467f",
              "status": "measured",
              "samples": [
                0.3,
                0.3,
                0.3
              ],
              "judge_sample_hashes": [
                "bf25bc6cf9555769774cd8fba5b16e0f1267fabdea005a217074c3a13be95e41",
                "f2cdf3bf9e21e23c1bb7deb248d73ed13f5193a50d3f956fc6dcc4de7f2d1b6b",
                "f54139f58a6f0a7a6ff421c04494cecfa8dcb1c293f07cf297a50ac25d3d1b47"
              ],
              "mean": 0.3,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 2553,
                "output_tokens": 259,
                "cached_tokens": 2053,
                "wall_ms": 6450
              },
              "judge_usage": {
                "input_tokens": 10197,
                "output_tokens": 373,
                "cached_tokens": 8953,
                "wall_ms": 12729
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "4757ce5f25fbbf883fcd5c6a8cd6a852ca671e87ea9942dedcfc24c4fe3585c0",
              "status": "measured",
              "samples": [
                0.3,
                0.3,
                0.3
              ],
              "judge_sample_hashes": [
                "9e6d4e5fae8fe39b5e3705daa35d525072081601120ffaf8dbb14c57d0349045",
                "240434fb58191266f5730cf8bd472a169d66a4fccd3dcfecc023bd3bbc5f7595",
                "44063d8c1c1d9defd1166064c7198a8c2cca230bdfdfd2199f02809bb9d7608f"
              ],
              "mean": 0.3,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 2553,
                "output_tokens": 483,
                "cached_tokens": 2551,
                "wall_ms": 9123
              },
              "judge_usage": {
                "input_tokens": 10524,
                "output_tokens": 312,
                "cached_tokens": 9171,
                "wall_ms": 12951
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "172e537d119953f63177b9d7492edc5c1ff56b8238068a01917e43a6b86a2e92",
              "status": "measured",
              "samples": [
                0.3,
                0.3,
                0.3
              ],
              "judge_sample_hashes": [
                "c3d73282997330f890f6353de483ce0e237c3cfdf68c009d2fb169132cb99799",
                "caaad1ee415feb18bb21f23c3a5e670e4a5327be04963f4f523e3a65ff5544c7",
                "7794a506e56c6dac5686b69669e5f5e1319a739ce0e95577fe1e9adfabe5b751"
              ],
              "mean": 0.3,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 2553,
                "output_tokens": 226,
                "cached_tokens": 2551,
                "wall_ms": 6020
              },
              "judge_usage": {
                "input_tokens": 10188,
                "output_tokens": 274,
                "cached_tokens": 8947,
                "wall_ms": 12425
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.862222,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.3,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.862222,
    "baseline_score": 0.3,
    "delta": 0.562222,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "2df794c7f2539a9ce4701a03997b667c90cef8c2136a77eabf39313df5b7af33",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 7557,
      "mean_output_tokens": 394.33,
      "mean_cost_usd_per_call": 0.047643,
      "median_wall_ms": 9001,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 2553,
      "mean_output_tokens": 322.67,
      "mean_cost_usd_per_call": 0.020832,
      "median_wall_ms": 6450,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.026811,
    "skill_incremental_cost_usd_per_1k_calls": 26.811,
    "output_tokens_delta": 71.66,
    "median_wall_ms_delta": 2551,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.123925,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}