{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-haiku-4-5-20251001",
    "model_release_date": "2025-10-01",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T21:50:37.187Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-4-5-20251001",
      "reported_models": [
        "claude-haiku-4-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T21:50:37.187Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-4-5-20251001": {
          "input_per_mtok": 1,
          "output_per_mtok": 5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.813333,
        "mean": 0.813333,
        "stddev": 0.003334,
        "samples": [
          0.84,
          0.83,
          0.76
        ],
        "generation_hash": "77c03e43806b191e574449deba642be007a83a541ce17db843cfe27cff70eb33",
        "judge_sample_hashes": [
          "49e5e5252e79b6ef23cccea45e858d5c27fa8b4414b5c6638cac2f4303b61c17",
          "4761141f09e9e745d458c399f9a9e51183802b17941ab52f07d3cdb313a77609",
          "a7994ead48861ed0a35679da540dc3cf9e43c847ed66dc5cc69ffb93e9e73326"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix:` prefix with a specific subject and a body stating both cause (wrong time unit) and user-facing impact (links expiring 24x early), slightly beyond baseline correctness.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43593,
          "output_tokens": 1770,
          "cached_tokens": 32317,
          "wall_ms": 33510
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.813333,
          "sd": 0.003334,
          "judge_sd_mean": 0.024713,
          "variance_ratio": 0.134909,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "94377932ce7fe20b3e662fe77083042aa9f2f150f95e6f66800329dffd2f9af9",
              "status": "measured",
              "samples": [
                0.8,
                0.82,
                0.83
              ],
              "judge_sample_hashes": [
                "1617411dee897cb10e87ea2e268b59bc571b68516e7a4dcdde976fb6dac9da79",
                "9ed926452ca07e16db5995e539a28ce60a89d46b32198195e1733cb750ded59d",
                "0c7013f50fccab9c34ea1993136ec74be9a12d7d1942387536e060de4de2aab7"
              ],
              "mean": 0.816667,
              "stddev": 0.015275,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 17698,
                "output_tokens": 1073,
                "cached_tokens": 17688,
                "wall_ms": 12726
              },
              "judge_usage": {
                "input_tokens": 43596,
                "output_tokens": 1056,
                "cached_tokens": 32319,
                "wall_ms": 99867
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "75b5e6f72d3f833a29817009eb828f98a17c1527cd7008069d122a69a5afdd6a",
              "status": "measured",
              "samples": [
                0.8,
                0.83,
                0.81
              ],
              "judge_sample_hashes": [
                "17b731738c9d4bc264fb869749e7bac7ee9e4570de5d28ee29e45d0055c0f0a4",
                "3baf6bd7de425c70e3f22a98a89097f86766ee345ec1fc2e5f8ec37f1f277751",
                "c8a1cee7cafbf2272c6e921d851d14fdce2034021fd78116967d4c153c8e7271"
              ],
              "mean": 0.813333,
              "stddev": 0.015275,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 17698,
                "output_tokens": 1048,
                "cached_tokens": 17688,
                "wall_ms": 12422
              },
              "judge_usage": {
                "input_tokens": 43629,
                "output_tokens": 994,
                "cached_tokens": 32341,
                "wall_ms": 20946
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "77c03e43806b191e574449deba642be007a83a541ce17db843cfe27cff70eb33",
              "status": "measured",
              "samples": [
                0.84,
                0.83,
                0.76
              ],
              "judge_sample_hashes": [
                "49e5e5252e79b6ef23cccea45e858d5c27fa8b4414b5c6638cac2f4303b61c17",
                "4761141f09e9e745d458c399f9a9e51183802b17941ab52f07d3cdb313a77609",
                "a7994ead48861ed0a35679da540dc3cf9e43c847ed66dc5cc69ffb93e9e73326"
              ],
              "mean": 0.81,
              "stddev": 0.043589,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 17698,
                "output_tokens": 1776,
                "cached_tokens": 17688,
                "wall_ms": 20164
              },
              "judge_usage": {
                "input_tokens": 43593,
                "output_tokens": 1770,
                "cached_tokens": 32317,
                "wall_ms": 33510
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.838889,
        "mean": 0.838889,
        "stddev": 0.013878,
        "samples": [
          0.85,
          0.84,
          0.84
        ],
        "generation_hash": "68c2fdc0f6abcc98ae262dc4de3d68e36449a6eb9edf1fe283eb3e324d5b6f3b",
        "judge_sample_hashes": [
          "df53dabf8fd4f936f82ac3db0045062406288c769a01e2a052be5b7120339268",
          "d5cab124e80a93757c700c45a857e26c312c69626b1146c580f8cac18637cb80",
          "5a3498b7b01c4aaa3efba933e15e954bb6ea0c50eec0f2a59d3461ffb1b6af7b"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix(auth):` prefix with a specific subject; body states user-facing symptom (~1h vs 24h) and the wrong-unit cause, exceeding a mere diff restatement.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43641,
          "output_tokens": 1101,
          "cached_tokens": 32349,
          "wall_ms": 31944
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.838889,
          "sd": 0.013878,
          "judge_sd_mean": 0.003849,
          "variance_ratio": 3.605612,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "d8bf8a58a28f083220f109cc2625d84ad71c19aa4f8056a253fc99f3551c018b",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "bf16b7e5ee5e9ed621e6058749ee4c08daa5756558dae7003ac3d895cf794d9a",
                "6d37d2786950a72db6555692d1417d7d17e40bab517a24ca3f52c8889d053880",
                "eda0d3b1eff30347c0a5aeac0eabc605c924f554c2a7aa105619c9a1a4ffba76"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14085,
                "output_tokens": 841,
                "cached_tokens": 14075,
                "wall_ms": 10845
              },
              "judge_usage": {
                "input_tokens": 43608,
                "output_tokens": 866,
                "cached_tokens": 32327,
                "wall_ms": 20839
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "328409e937f3de2c6f42d188eae2c854d055821c5aeece22a0c31d792a7a8756",
              "status": "measured",
              "samples": [
                0.82,
                0.83,
                0.82
              ],
              "judge_sample_hashes": [
                "cb43e20b50feb256dcea4e1e5a0e6a73304f6c124c0813260028524bcf9e7661",
                "17a30e392dd5f9f6f5510ca956bd5b1bdbddbf2f0fdd387cf0cfea6831bf77f4",
                "84118b2add5cd5f98da8da84dfdbd64bd87c2fe5f4fd6003227c4a3a15b40783"
              ],
              "mean": 0.823333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14085,
                "output_tokens": 754,
                "cached_tokens": 14075,
                "wall_ms": 9590
              },
              "judge_usage": {
                "input_tokens": 43563,
                "output_tokens": 1106,
                "cached_tokens": 32297,
                "wall_ms": 24765
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "68c2fdc0f6abcc98ae262dc4de3d68e36449a6eb9edf1fe283eb3e324d5b6f3b",
              "status": "measured",
              "samples": [
                0.85,
                0.84,
                0.84
              ],
              "judge_sample_hashes": [
                "df53dabf8fd4f936f82ac3db0045062406288c769a01e2a052be5b7120339268",
                "d5cab124e80a93757c700c45a857e26c312c69626b1146c580f8cac18637cb80",
                "5a3498b7b01c4aaa3efba933e15e954bb6ea0c50eec0f2a59d3461ffb1b6af7b"
              ],
              "mean": 0.843333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14085,
                "output_tokens": 601,
                "cached_tokens": 14075,
                "wall_ms": 7285
              },
              "judge_usage": {
                "input_tokens": 43641,
                "output_tokens": 1101,
                "cached_tokens": 32349,
                "wall_ms": 31944
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.813333,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.838889,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.813333,
    "baseline_score": 0.838889,
    "delta": -0.025556,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "bf1e17919f1e906859cc53120fd6f949f06c71bddd8f198e463792de623c0602",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 17698,
      "mean_output_tokens": 1299,
      "mean_cost_usd_per_call": 0.024193,
      "median_wall_ms": 12726,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 14085,
      "mean_output_tokens": 732,
      "mean_cost_usd_per_call": 0.017745,
      "median_wall_ms": 9590,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.006448,
    "skill_incremental_cost_usd_per_1k_calls": 6.448,
    "output_tokens_delta": 567,
    "median_wall_ms_delta": 3136,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.507945,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}