{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-sonnet-5",
    "model_release_date": "2026-06-30",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T04:58:46.174Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T04:58:46.174Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.851111,
        "mean": 0.851111,
        "stddev": 0.001924,
        "samples": [
          0.85,
          0.85,
          0.85
        ],
        "generation_hash": "d50c6610d39889920895fb17a37262099fc0fa3419e79da3e066107fb6ec06f7",
        "judge_sample_hashes": [
          "d8793c38b8aa37553fab280289c8dba78aa6eea8c2318a01390661a6c3f94abb",
          "0a4ee6af19b61148a9cffd583e685052c6859db05d18770e90ca8d424732bc28",
          "6178152a2f5f347ad8be567ec5e3963bdca233f8ce581bd093643620e0eb1908"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix:` prefix with a specific subject and a body conveying user-facing impact (1h vs 24h) and cause (missing unit conversion); slightly speculative millisecond detail prevents higher.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43707,
          "output_tokens": 1022,
          "cached_tokens": 32389,
          "wall_ms": 23579
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.851111,
          "sd": 0.001924,
          "judge_sd_mean": 0.001925,
          "variance_ratio": 0.999481,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "16d0f964b24dff4f414449b391387b6a07501e68d7b2b1da070304f7970fe631",
              "status": "measured",
              "samples": [
                0.85,
                0.86,
                0.85
              ],
              "judge_sample_hashes": [
                "fb68dee0216428d97739a523c6dcef4caeccf6920680a9899a34a6af9ac30cd4",
                "e6fe79594c2d4da303df2aaa986a5f774fee3bab986889074ab8cec34c76b561",
                "2204f82bcc4d1cd1d6cbdaaae5c8d449f4cad9984cc21ce24840439c17ccd187"
              ],
              "mean": 0.853333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 24155,
                "output_tokens": 110,
                "cached_tokens": 24153,
                "wall_ms": 4453
              },
              "judge_usage": {
                "input_tokens": 43647,
                "output_tokens": 867,
                "cached_tokens": 32349,
                "wall_ms": 19723
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "46c868c9fa005d77b417199529b562eeab78e681c7c89f7bb2edb0a60cc2ef35",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "c09f48c268415a724a4572afe775edc6408cab1c4a1f43adb934a3e458694739",
                "f421c63c709538d0c07e66105a1d237283dcee38c27833fd0c3cfc9ef5a77333",
                "42cd7432a2e3d0f1af1a4e04e75de3ebce20e434cc8c1728bbe1b2cee5fc630c"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 24155,
                "output_tokens": 79,
                "cached_tokens": 24153,
                "wall_ms": 4304
              },
              "judge_usage": {
                "input_tokens": 43554,
                "output_tokens": 875,
                "cached_tokens": 32287,
                "wall_ms": 23215
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "d50c6610d39889920895fb17a37262099fc0fa3419e79da3e066107fb6ec06f7",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "d8793c38b8aa37553fab280289c8dba78aa6eea8c2318a01390661a6c3f94abb",
                "0a4ee6af19b61148a9cffd583e685052c6859db05d18770e90ca8d424732bc28",
                "6178152a2f5f347ad8be567ec5e3963bdca233f8ce581bd093643620e0eb1908"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 24155,
                "output_tokens": 130,
                "cached_tokens": 24153,
                "wall_ms": 4556
              },
              "judge_usage": {
                "input_tokens": 43707,
                "output_tokens": 1022,
                "cached_tokens": 32389,
                "wall_ms": 23579
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.831111,
        "mean": 0.831111,
        "stddev": 0.032717,
        "samples": [
          0.85,
          0.85,
          0.85
        ],
        "generation_hash": "27ac44ae1ac6767f24bbbbb0341730b4f88a9193ee7b684ddffd43c5cbef6817",
        "judge_sample_hashes": [
          "82991b2134f8b58e232eb8d70d05a9c6b7bec527bd014152596c0f6c527953e0",
          "99d0ba3fe8c0b30086679ac1cf0f066a26643fb670a255a9e19373f56cee103d",
          "918af210715e27479e4b5a44e138d4ea0a8f2250594b4770d0c81931e08cdf54"
        ],
        "threshold": 0.7,
        "reason": "Conventional `fix(auth):` prefix, specific subject, and body conveying both user-facing impact (~1h vs 24h) and root cause (wrong time unit); actionable for maintainers.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43602,
          "output_tokens": 805,
          "cached_tokens": 32319,
          "wall_ms": 19342
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.831111,
          "sd": 0.032717,
          "judge_sd_mean": 0.003849,
          "variance_ratio": 8.50013,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "3fa2ed51d34b19ee09c06ebb9408e6f04e02c4cd46a9ba5097482b33e58f90ed",
              "status": "measured",
              "samples": [
                0.8,
                0.78,
                0.8
              ],
              "judge_sample_hashes": [
                "10d33bf4ba2cba2659082e09d07fd88199ff79291ec90156095f96b457af7861",
                "c21a85a1e740dd823ac75a1859c8cde37a9b46aa24e1104aa31d339e544e4c57",
                "306b4935c4186ac824a2ac1fa1b0192c02ebc0a7d43811f2f87b1f8e7239b5ff"
              ],
              "mean": 0.793333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19151,
                "output_tokens": 127,
                "cached_tokens": 19149,
                "wall_ms": 4627
              },
              "judge_usage": {
                "input_tokens": 43698,
                "output_tokens": 1005,
                "cached_tokens": 32383,
                "wall_ms": 20820
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "175a5ad7bb81a05de50c1bef902338b14e653bbbf15ff751c3c8071a25ee3a84",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "8c973c7742d864c7a402eac1347b2b953dafc9061fcf586b475d8f13c511f314",
                "eaa9157dc1f917720c1de564251ec5d07a710f86eee3b066b9b480100c543505",
                "c3f5d61cd7d53d99542a9d67ba4b4030ee3c5df7c8fd35519d91f2f673cc3e60"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19151,
                "output_tokens": 93,
                "cached_tokens": 19149,
                "wall_ms": 4367
              },
              "judge_usage": {
                "input_tokens": 43596,
                "output_tokens": 821,
                "cached_tokens": 32315,
                "wall_ms": 19905
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "27ac44ae1ac6767f24bbbbb0341730b4f88a9193ee7b684ddffd43c5cbef6817",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "82991b2134f8b58e232eb8d70d05a9c6b7bec527bd014152596c0f6c527953e0",
                "99d0ba3fe8c0b30086679ac1cf0f066a26643fb670a255a9e19373f56cee103d",
                "918af210715e27479e4b5a44e138d4ea0a8f2250594b4770d0c81931e08cdf54"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19151,
                "output_tokens": 95,
                "cached_tokens": 19149,
                "wall_ms": 3152
              },
              "judge_usage": {
                "input_tokens": 43602,
                "output_tokens": 805,
                "cached_tokens": 32319,
                "wall_ms": 19342
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.851111,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.831111,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.851111,
    "baseline_score": 0.831111,
    "delta": 0.02,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "8a6cf114ed244626e8806ffb94597041f06b7401d1e425226318f510eea24e8e",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 24155,
      "mean_output_tokens": 106.33,
      "mean_cost_usd_per_call": 0.049373,
      "median_wall_ms": 4453,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 19151,
      "mean_output_tokens": 105,
      "mean_cost_usd_per_call": 0.039352,
      "median_wall_ms": 4367,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.010021,
    "skill_incremental_cost_usd_per_1k_calls": 10.021,
    "output_tokens_delta": 1.33,
    "median_wall_ms_delta": 86,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.48222,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}