{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-sonnet-5",
    "model_release_date": "2026-06-30",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T06:00:53.481Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T06:00:53.481Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.831111,
        "mean": 0.831111,
        "stddev": 0.010183,
        "samples": [
          0.82,
          0.84,
          0.84
        ],
        "generation_hash": "ba2da8ed21d37a1bf91dcd9a606d4541ab24e00add67d22e61834dbefe217c3a",
        "judge_sample_hashes": [
          "d499a5c18d52aa4c1821da71a6c5db199066227b0d87fcecee9b92b0f48e2c77",
          "95a54f2da7e55d6bf60f1147d8e105c6182271434aa47b3d4aabda49ecd5fa6b",
          "ebae91c73b27ff57b60d45fc631ac9893db422092ed4a38c42280f32d955cdb0"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix:` prefix with a specific subject and a body giving cause (wrong unit) and impact (~1h vs 24h); cause arithmetic slightly inconsistent, limiting exemplary credit.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43644,
          "output_tokens": 1309,
          "cached_tokens": 32347,
          "wall_ms": 24925
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.831111,
          "sd": 0.010183,
          "judge_sd_mean": 0.016289,
          "variance_ratio": 0.625146,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "8add38408a6d5996a5016640915e7821c114fa2bb0cfabb56dcccdcf3c53a5a7",
              "status": "measured",
              "samples": [
                0.82,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "a9d6c41faa7e57b3d9fb8d888a6092b410e11f1d5efc38e4eea4c5489874510c",
                "d0add33465ecf0746b380c62b3d3d46c2a90ae8d88771a7ca65ea42dae436266",
                "5f63c3279957adfecd5f122deb8dc6d5dbab325585108554e690ed8914533448"
              ],
              "mean": 0.84,
              "stddev": 0.017321,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 24155,
                "output_tokens": 90,
                "cached_tokens": 3406,
                "wall_ms": 4529
              },
              "judge_usage": {
                "input_tokens": 43587,
                "output_tokens": 982,
                "cached_tokens": 32309,
                "wall_ms": 21210
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "906bc04a24fe58c163786d4492b133039250f43c018ce542ed6cc1992cc195a7",
              "status": "measured",
              "samples": [
                0.84,
                0.8,
                0.82
              ],
              "judge_sample_hashes": [
                "bbf5ece5a72c2022191db0a1f935286f2da709bd1b5979e4176cd031e80b252d",
                "c018607029b1a57ce9d2f96f0d7bc0b4e5609da76521bb6cd62d64df6b8db01f",
                "5486c308e88b884f5f00300482ebe8014c567eed5db9057a6524c160ea2ed214"
              ],
              "mean": 0.82,
              "stddev": 0.02,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 24155,
                "output_tokens": 82,
                "cached_tokens": 24153,
                "wall_ms": 3338
              },
              "judge_usage": {
                "input_tokens": 43563,
                "output_tokens": 1378,
                "cached_tokens": 32293,
                "wall_ms": 26917
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "ba2da8ed21d37a1bf91dcd9a606d4541ab24e00add67d22e61834dbefe217c3a",
              "status": "measured",
              "samples": [
                0.82,
                0.84,
                0.84
              ],
              "judge_sample_hashes": [
                "d499a5c18d52aa4c1821da71a6c5db199066227b0d87fcecee9b92b0f48e2c77",
                "95a54f2da7e55d6bf60f1147d8e105c6182271434aa47b3d4aabda49ecd5fa6b",
                "ebae91c73b27ff57b60d45fc631ac9893db422092ed4a38c42280f32d955cdb0"
              ],
              "mean": 0.833333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 24155,
                "output_tokens": 109,
                "cached_tokens": 24153,
                "wall_ms": 4623
              },
              "judge_usage": {
                "input_tokens": 43644,
                "output_tokens": 1309,
                "cached_tokens": 32347,
                "wall_ms": 24925
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.835556,
        "mean": 0.835556,
        "stddev": 0.025239,
        "samples": [
          0.8,
          0.82,
          0.8
        ],
        "generation_hash": "1819f034d5406a05bf7fa0bd98df7dc4812fadf8f156fa877765e23b5a12e61a",
        "judge_sample_hashes": [
          "bcdcb6eed59d063d08867039b3f5134acdb7fcb5dd75d55d925417fbca44aca2",
          "a5c1d0ab998efedd04d6fe1f26094da6e23646e9243b9fd57eab1268b4c5d740",
          "62d7470533eabb975578b38c34a01f5a01cdf0517b20aae87c5dccde0b32af6a"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix(auth):` prefix, specific subject, and a body giving cause and user-facing impact; invented minutes-as-milliseconds detail is arithmetically inconsistent with ~1 hour, so not exemplary.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43635,
          "output_tokens": 1878,
          "cached_tokens": 32341,
          "wall_ms": 32320
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.835556,
          "sd": 0.025239,
          "judge_sd_mean": 0.007698,
          "variance_ratio": 3.278644,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "db2f124f63500cf718af265464dc5bc15ace17d54aed44e6bb47c6794563192c",
              "status": "measured",
              "samples": [
                0.86,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "67cdaf0b105dddc80ff27f144699951e6f5e1d6da145c4430c2e64486f276f80",
                "445ad286c12447d9c6b50a3aad43c623ed9e29d8360861bde7a9737dd9991c9f",
                "d20c5af1d63f881be5c60d6fffa1b2be702afde52cef1cd8e7869e2172fe14b2"
              ],
              "mean": 0.853333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19151,
                "output_tokens": 94,
                "cached_tokens": 8503,
                "wall_ms": 3250
              },
              "judge_usage": {
                "input_tokens": 43599,
                "output_tokens": 775,
                "cached_tokens": 32317,
                "wall_ms": 19709
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "6887802ca17a5b1deacbafa51e9b44dbecf04a837b7a2be0d86ad80ce635a071",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.84
              ],
              "judge_sample_hashes": [
                "58160de59f25420ee2fa1257bf629290e0fa9fa40e5a67ec262558b0f4f73328",
                "b66f4501779475df33942d3a8eb9fb8c910d05b651ef05825ebb1827bb6d92ca",
                "631e0683542bfcda04e5e4149db15b7049886b85386d4fb471eba6ee765f6be1"
              ],
              "mean": 0.846667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19151,
                "output_tokens": 141,
                "cached_tokens": 19149,
                "wall_ms": 5177
              },
              "judge_usage": {
                "input_tokens": 43740,
                "output_tokens": 1018,
                "cached_tokens": 32411,
                "wall_ms": 30262
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "1819f034d5406a05bf7fa0bd98df7dc4812fadf8f156fa877765e23b5a12e61a",
              "status": "measured",
              "samples": [
                0.8,
                0.82,
                0.8
              ],
              "judge_sample_hashes": [
                "bcdcb6eed59d063d08867039b3f5134acdb7fcb5dd75d55d925417fbca44aca2",
                "a5c1d0ab998efedd04d6fe1f26094da6e23646e9243b9fd57eab1268b4c5d740",
                "62d7470533eabb975578b38c34a01f5a01cdf0517b20aae87c5dccde0b32af6a"
              ],
              "mean": 0.806667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19151,
                "output_tokens": 106,
                "cached_tokens": 19149,
                "wall_ms": 5510
              },
              "judge_usage": {
                "input_tokens": 43635,
                "output_tokens": 1878,
                "cached_tokens": 32341,
                "wall_ms": 32320
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.831111,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.835556,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.831111,
    "baseline_score": 0.835556,
    "delta": -0.004445,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "f3c90a5948a5dd76a027d8e165e52afa6be3d10d70ee146b6f30e628662e9c00",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 24155,
      "mean_output_tokens": 93.67,
      "mean_cost_usd_per_call": 0.049247,
      "median_wall_ms": 4529,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 19151,
      "mean_output_tokens": 113.67,
      "mean_cost_usd_per_call": 0.039439,
      "median_wall_ms": 5177,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.009808,
    "skill_incremental_cost_usd_per_1k_calls": 9.808,
    "output_tokens_delta": -20,
    "median_wall_ms_delta": -648,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.51607,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}