{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-sonnet-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T04:00:32.606Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T04:00:32.606Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.86,
        "mean": 0.86,
        "stddev": 0.008819,
        "samples": [
          0.87,
          0.86,
          0.86
        ],
        "generation_hash": "27c550ded2a0ddd35737a046fea29607b00398233d428281160d480d71798bb9",
        "judge_sample_hashes": [
          "79a1e15319764c4df9555f23370d2f4ae1401bfd084f0d6407d0f92f249b8dfc",
          "b7c76260a845c8a13cab4fd66a53d9e4bd40c1f86522a912c48d91609a6e89eb",
          "c7c3712c6311ce0fa81515f0115da5c5d365fffa69cba76db538c14c6dd4fdff"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix:` prefix with specific subject; body explains wrong-unit cause and user-facing impact (expired links/error) plus intent, exceeding a bare diff restatement.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43746,
          "output_tokens": 720,
          "cached_tokens": 32415,
          "wall_ms": 21858
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.86,
          "sd": 0.008819,
          "judge_sd_mean": 0.003849,
          "variance_ratio": 2.291244,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "1454ec83e479e398b48716ed7e02736818482f2b02b090d801efda7a291dd1f3",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "d9a2256b681561f24554686cb5955ef87ff1993dcc217ab16f804fa146864787",
                "712728061aa1b735d13490af95f30a2f0a04881a08431e488ee4b8c91222adb9",
                "c1f8ba24965656171930c98a0cfca4eef2ceade76bcaa5578df3bee0c04a6dae"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 17737,
                "output_tokens": 233,
                "cached_tokens": 540,
                "wall_ms": 4319
              },
              "judge_usage": {
                "input_tokens": 44016,
                "output_tokens": 1017,
                "cached_tokens": 32595,
                "wall_ms": 20682
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "bc9c25522d627f5eea2ca44808625ba427050b6ff74dca5b7c6274ecf1d4ccba",
              "status": "measured",
              "samples": [
                0.86,
                0.87,
                0.87
              ],
              "judge_sample_hashes": [
                "3b710ecfc9669d447b0d47f55c2cf255c68cbb107b0673a273ba7fd4ed62c9b2",
                "f66ec7d7124b09fdca201386bed411db3337e9a416a2db538c5edc013762f3bb",
                "884b63e504fdce5742870774538e9f586c411e1744a784d3b0096fe822eeff0f"
              ],
              "mean": 0.866667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 17737,
                "output_tokens": 172,
                "cached_tokens": 17735,
                "wall_ms": 3967
              },
              "judge_usage": {
                "input_tokens": 43833,
                "output_tokens": 757,
                "cached_tokens": 32473,
                "wall_ms": 19468
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "27c550ded2a0ddd35737a046fea29607b00398233d428281160d480d71798bb9",
              "status": "measured",
              "samples": [
                0.87,
                0.86,
                0.86
              ],
              "judge_sample_hashes": [
                "79a1e15319764c4df9555f23370d2f4ae1401bfd084f0d6407d0f92f249b8dfc",
                "b7c76260a845c8a13cab4fd66a53d9e4bd40c1f86522a912c48d91609a6e89eb",
                "c7c3712c6311ce0fa81515f0115da5c5d365fffa69cba76db538c14c6dd4fdff"
              ],
              "mean": 0.863333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 17737,
                "output_tokens": 143,
                "cached_tokens": 17735,
                "wall_ms": 5415
              },
              "judge_usage": {
                "input_tokens": 43746,
                "output_tokens": 720,
                "cached_tokens": 32415,
                "wall_ms": 21858
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.86,
        "mean": 0.86,
        "stddev": 0.008819,
        "samples": [
          0.87,
          0.87,
          0.86
        ],
        "generation_hash": "979684c861f4f2b420a84679c063dc75ecac9a31bd509085d3c0aef7857db1e8",
        "judge_sample_hashes": [
          "1e5e9772f7988b0e5ccbcda43b59e0b293ee02fa242d18467ce51576d0e90426",
          "cb2bae1f262c6fbe72ff2e76344be7a2c943f883223f4d03d4d199e6fcbca3ab",
          "23ffba7d63ac61e9aff4db594f45d7228ab4494bf79d9ca6c54d7397488b7765"
        ],
        "threshold": 0.7,
        "reason": "Subject uses `fix(auth):` with a specific description; body explains the user-facing symptom, the unit-conversion cause, the fix, and already-issued-link impact — exemplary but not flawless.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43872,
          "output_tokens": 814,
          "cached_tokens": 32499,
          "wall_ms": 22536
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.86,
          "sd": 0.008819,
          "judge_sd_mean": 0.003849,
          "variance_ratio": 2.291244,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "10e2519949567f14cba0b983f93192224e38409def2fffe96430525a9d78e8ff",
              "status": "measured",
              "samples": [
                0.86,
                0.86,
                0.87
              ],
              "judge_sample_hashes": [
                "e00968019871765e58ef139166d32fd13aa9d9b92114d1cc6b8c638bbf422868",
                "95a4c31a641e459b2a4951801599b1405dea342785ffd159d14b091cd7b9c668",
                "530d401a77a825420ff7d7da9a134cd83e0ea89c6e70311d261b83041025e07d"
              ],
              "mean": 0.863333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12733,
                "output_tokens": 187,
                "cached_tokens": 2144,
                "wall_ms": 3528
              },
              "judge_usage": {
                "input_tokens": 43878,
                "output_tokens": 650,
                "cached_tokens": 32503,
                "wall_ms": 18468
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "70a034c5e8b176591967dd20ac2c7480c49fb8ec0462cf2bf6ed22b35c35b0b7",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "beed0b35be8be8c9b711df575339c77ed2e7152f2300600ffd528d296775c64d",
                "8ff5d9b459f777dd65e820f40312544c4c09b9128c8dd25ce5cc703249e21317",
                "c1ddc78720cf6d34037a907ce079fde466f0ae821de5ede3bf2cfb0b9f1f0d10"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12733,
                "output_tokens": 153,
                "cached_tokens": 12731,
                "wall_ms": 6986
              },
              "judge_usage": {
                "input_tokens": 43776,
                "output_tokens": 942,
                "cached_tokens": 32435,
                "wall_ms": 20922
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "979684c861f4f2b420a84679c063dc75ecac9a31bd509085d3c0aef7857db1e8",
              "status": "measured",
              "samples": [
                0.87,
                0.87,
                0.86
              ],
              "judge_sample_hashes": [
                "1e5e9772f7988b0e5ccbcda43b59e0b293ee02fa242d18467ce51576d0e90426",
                "cb2bae1f262c6fbe72ff2e76344be7a2c943f883223f4d03d4d199e6fcbca3ab",
                "23ffba7d63ac61e9aff4db594f45d7228ab4494bf79d9ca6c54d7397488b7765"
              ],
              "mean": 0.866667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12733,
                "output_tokens": 185,
                "cached_tokens": 12731,
                "wall_ms": 4696
              },
              "judge_usage": {
                "input_tokens": 43872,
                "output_tokens": 814,
                "cached_tokens": 32499,
                "wall_ms": 22536
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.86,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.86,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.86,
    "baseline_score": 0.86,
    "delta": 0,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "8842319fabc6851009e75c3c49031ecadc35235a873119e5f7dd385d06460160",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 17737,
      "mean_output_tokens": 182.67,
      "mean_cost_usd_per_call": 0.037301,
      "median_wall_ms": 4319,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 12733,
      "mean_output_tokens": 175,
      "mean_cost_usd_per_call": 0.027216,
      "median_wall_ms": 4696,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.010085,
    "skill_incremental_cost_usd_per_1k_calls": 10.085,
    "output_tokens_delta": 7.67,
    "median_wall_ms_delta": -377,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.47644,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}