{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-haiku-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T20:29:31.013Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-5-5",
      "reported_models": [
        "claude-haiku-5-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T20:29:31.013Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-5-5": {
          "input_per_mtok": 0.1,
          "output_per_mtok": 0.5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.848889,
        "mean": 0.848889,
        "stddev": 0.007698,
        "samples": [
          0.84,
          0.84,
          0.84
        ],
        "generation_hash": "3fc2ece0b738b67629e2bfedd1f9fcc4f4b4f971150a4b0c477ad243faa1161f",
        "judge_sample_hashes": [
          "0ea6651dfd7345f145fabb401bb27d17c3cc1b08179129269724f69be95fad0c",
          "e37925bc090cc9238bae3ee7f3009206baf5d20e14216934f6cf365a78307835",
          "41d9f56ed16029982ddc3eafff4677d00d33f079a43947285d55497faeb9c0df"
        ],
        "threshold": 0.7,
        "reason": "Uses conventional `fix(auth):` prefix with a specific subject, and the body conveys both the user-facing symptom (1h vs 24h expiry) and the wrong-unit cause; slightly redundant phrasing.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43659,
          "output_tokens": 929,
          "cached_tokens": 32361,
          "wall_ms": 19727
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.848889,
          "sd": 0.007698,
          "judge_sd_mean": 0.003849,
          "variance_ratio": 2,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "22111ea019d6d6322c38ca70dc88a2c71c3bd3d348b0f7657ee84b1b8aae2b67",
              "status": "measured",
              "samples": [
                0.86,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "8a5301c4dfa78ba190f9f90d3a540aa33be9a6bc13b41c33244256ec5bb94c15",
                "305851761d6cf2511cbe9ead0adcdeed879b43e2ce97fc0f50c7733027f6bca6",
                "738787a5683b9963389f29ba750b0a0477f5cc6c011f7b6efc3478d21ca1845f"
              ],
              "mean": 0.853333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 24834,
                "output_tokens": 696,
                "cached_tokens": 3396,
                "wall_ms": 5869
              },
              "judge_usage": {
                "input_tokens": 43716,
                "output_tokens": 736,
                "cached_tokens": 32399,
                "wall_ms": 30178
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "250916c35cd3156fb998384057aa73a2125b5c76bb9791e5e12704e5b944940d",
              "status": "measured",
              "samples": [
                0.86,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "d2e91918cdfc33e64a90760f50dc4a365fb76f36ecf23daa8a0f0bdbf253af9e",
                "3509e0b7e5a8e795d2ac73be86b5bec96c844ce54d02d03245762d8f889356e4",
                "ab839bc48ac39c0f1590a2bd56373c93e05b13ede524a39997d17875f05459a0"
              ],
              "mean": 0.853333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 24834,
                "output_tokens": 538,
                "cached_tokens": 24832,
                "wall_ms": 5194
              },
              "judge_usage": {
                "input_tokens": 43599,
                "output_tokens": 712,
                "cached_tokens": 32321,
                "wall_ms": 43314
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "3fc2ece0b738b67629e2bfedd1f9fcc4f4b4f971150a4b0c477ad243faa1161f",
              "status": "measured",
              "samples": [
                0.84,
                0.84,
                0.84
              ],
              "judge_sample_hashes": [
                "0ea6651dfd7345f145fabb401bb27d17c3cc1b08179129269724f69be95fad0c",
                "e37925bc090cc9238bae3ee7f3009206baf5d20e14216934f6cf365a78307835",
                "41d9f56ed16029982ddc3eafff4677d00d33f079a43947285d55497faeb9c0df"
              ],
              "mean": 0.84,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 24834,
                "output_tokens": 575,
                "cached_tokens": 24832,
                "wall_ms": 5412
              },
              "judge_usage": {
                "input_tokens": 43659,
                "output_tokens": 929,
                "cached_tokens": 32361,
                "wall_ms": 19727
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.3,
        "mean": 0.3,
        "stddev": 0,
        "samples": [
          0.3,
          0.3,
          0.3
        ],
        "generation_hash": "2d863410237b2e051c0977dad0c0de9f44aae2896d68cb6a924a2217c507ae08",
        "judge_sample_hashes": [
          "28b864d40b8973dbb63540be1e4d72f8fdc880b6582edb0d69ace2ea67cc870f",
          "f6e186d9c64c20778be94807193a7bcae510126be80bd9175d5c402576514f0f",
          "25e33fa8c63d6a367d770155a5073100a0da054fe4838c6beee4156ec653caf8"
        ],
        "threshold": 0.7,
        "reason": "Subject reads 'Fix password-reset link expiry...' without the conventional `fix:` prefix, triggering the 0.3 cap despite a specific subject and a clear why-focused body.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43917,
          "output_tokens": 689,
          "cached_tokens": 32533,
          "wall_ms": 29870
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.3,
          "sd": 0,
          "judge_sd_mean": 0,
          "variance_ratio": null,
          "variance_ratio_unavailable": "judge_sd_zero",
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "be963958787e733d17dac4f4c3c9626710b44745fefce62d502de6d9cb3a079b",
              "status": "measured",
              "samples": [
                0.3,
                0.3,
                0.3
              ],
              "judge_sample_hashes": [
                "c79f006fa5f8b06dd5e5f6fcd3f1070cc48ae70773dcaa23668bafd868247a9b",
                "5658a1a1773698d2d74cd77d703a53ee8af3e907a9345d60868afb5dd13216a6",
                "be19b575475eb78666106d85362d052352b08d161d360d314c9894db1a96b089"
              ],
              "mean": 0.3,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19830,
                "output_tokens": 358,
                "cached_tokens": 9273,
                "wall_ms": 4046
              },
              "judge_usage": {
                "input_tokens": 43788,
                "output_tokens": 664,
                "cached_tokens": 32447,
                "wall_ms": 37596
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "c988b6e102a0dac011be210f1523c700fa56c3a0eb30e8204ccdf4482c9a44d5",
              "status": "measured",
              "samples": [
                0.3,
                0.3,
                0.3
              ],
              "judge_sample_hashes": [
                "4b0b4c890c0b1fb0d84601e25f05ad4a62a23d929adf37bf3a6fc3ad21b94288",
                "bb538c9b558296f216a50909f72226363982c317e09906b6b8f76b1aaf305bbc",
                "e0dd3524be0a54cb39dc89a80018a4fbde1b62f5254a64d619d22bea92c28f4e"
              ],
              "mean": 0.3,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19830,
                "output_tokens": 769,
                "cached_tokens": 19828,
                "wall_ms": 6059
              },
              "judge_usage": {
                "input_tokens": 43938,
                "output_tokens": 595,
                "cached_tokens": 32547,
                "wall_ms": 39528
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "2d863410237b2e051c0977dad0c0de9f44aae2896d68cb6a924a2217c507ae08",
              "status": "measured",
              "samples": [
                0.3,
                0.3,
                0.3
              ],
              "judge_sample_hashes": [
                "28b864d40b8973dbb63540be1e4d72f8fdc880b6582edb0d69ace2ea67cc870f",
                "f6e186d9c64c20778be94807193a7bcae510126be80bd9175d5c402576514f0f",
                "25e33fa8c63d6a367d770155a5073100a0da054fe4838c6beee4156ec653caf8"
              ],
              "mean": 0.3,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19830,
                "output_tokens": 958,
                "cached_tokens": 19828,
                "wall_ms": 6986
              },
              "judge_usage": {
                "input_tokens": 43917,
                "output_tokens": 689,
                "cached_tokens": 32533,
                "wall_ms": 29870
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.848889,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.3,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.848889,
    "baseline_score": 0.3,
    "delta": 0.548889,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "8b2539f6940219311968f6831ab6a51f87125f1ec2e9dc8b8047bec1393390fd",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 24834,
      "mean_output_tokens": 603,
      "mean_cost_usd_per_call": 0.002785,
      "median_wall_ms": 5412,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 19830,
      "mean_output_tokens": 695,
      "mean_cost_usd_per_call": 0.002331,
      "median_wall_ms": 6059,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.000454,
    "skill_incremental_cost_usd_per_1k_calls": 0.454,
    "output_tokens_delta": -92,
    "median_wall_ms_delta": -647,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.47833,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}