{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-opus-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-23T06:36:59.519Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-opus-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-23T06:36:59.519Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5-5": {
          "input_per_mtok": 4,
          "output_per_mtok": 20,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.837778,
        "mean": 0.837778,
        "stddev": 0.027148,
        "samples": [
          0.85,
          0.85,
          0.85
        ],
        "generation_hash": "3b509bb3783534439af164b4e2838109f7491289370a365eb56b261a5dc87e6c",
        "judge_sample_hashes": [
          "a30470a438a884b43efaec7cab6c69707eea0481348c80e9cc45bc13b47ad08b",
          "93668a1f121068679b8d522cc439acceff3fbc24f2a0853fadac75c640bd43fb",
          "8fe5a4ac8ebd5b8dba8651c9a6e5b3936c18f3ec1f00c75f16b7aed26f004b1e"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix:` prefix, specific subject, and a body conveying both user-facing impact (early expiry, forced re-request) and cause (missing unit conversion) beyond the diff.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 16749,
          "output_tokens": 949,
          "cached_tokens": 14418,
          "wall_ms": 18247
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.837778,
          "sd": 0.027148,
          "judge_sd_mean": 0.007698,
          "variance_ratio": 3.52663,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "9b70bbce7986d8f1f0d5962182fcb440797f49a9f849716a8941d0cc74791c26",
              "status": "measured",
              "samples": [
                0.8,
                0.82,
                0.8
              ],
              "judge_sample_hashes": [
                "747e17c5c4ad26ff717b09881760323f79a6498a9551cdbe1894824e0066e39c",
                "399e685450a9584c06993b22145e39571f63825b8f1dd128d793601b5660162b",
                "fa13f806675ea393a61b1bd78b9acd736c4ef9c83cb1ddad8f13ecfd85dfc362"
              ],
              "mean": 0.806667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 8667,
                "output_tokens": 228,
                "cached_tokens": 540,
                "wall_ms": 5457
              },
              "judge_usage": {
                "input_tokens": 16803,
                "output_tokens": 1288,
                "cached_tokens": 14454,
                "wall_ms": 22204
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "66a0d990e1f7d747975187b304d4d0842ab50d69c9393489362cf5c08b2e434b",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.87
              ],
              "judge_sample_hashes": [
                "46b1f98131229d43c2b86cccfd1a7143c867a6f35fbc14223f644c9b655cbd6b",
                "c32a53afc12692b3eb26d3500d86193d11137b7793b2762740636366f424dcf8",
                "63badd01299edc245ceec9d58cecad20a10acd4087d2e20bda2983926cddb9ef"
              ],
              "mean": 0.856667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 8667,
                "output_tokens": 342,
                "cached_tokens": 8665,
                "wall_ms": 6273
              },
              "judge_usage": {
                "input_tokens": 16632,
                "output_tokens": 869,
                "cached_tokens": 14340,
                "wall_ms": 19708
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "3b509bb3783534439af164b4e2838109f7491289370a365eb56b261a5dc87e6c",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "a30470a438a884b43efaec7cab6c69707eea0481348c80e9cc45bc13b47ad08b",
                "93668a1f121068679b8d522cc439acceff3fbc24f2a0853fadac75c640bd43fb",
                "8fe5a4ac8ebd5b8dba8651c9a6e5b3936c18f3ec1f00c75f16b7aed26f004b1e"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 8667,
                "output_tokens": 307,
                "cached_tokens": 8665,
                "wall_ms": 6110
              },
              "judge_usage": {
                "input_tokens": 16749,
                "output_tokens": 949,
                "cached_tokens": 14418,
                "wall_ms": 18247
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.3,
        "mean": 0.3,
        "stddev": 0,
        "samples": [
          0.3,
          0.3,
          0.3
        ],
        "generation_hash": "4f8747d1b186626d570c87ea223d75f840f2186675fd13a0127b90abfab53615",
        "judge_sample_hashes": [
          "97205b31d50757057faa21bd165e0524ae703237ac8f896870f616af9bda0617",
          "07f025bdad47f5fed4ead9cb780feab3954a1da8f8616d4a429ac3ea85f3a38d",
          "f046b5fbf5c648305acca07a91418828e357bdfbfcfd78358e84ed93ba272380"
        ],
        "threshold": 0.7,
        "reason": "Body strongly explains cause and user-facing impact, and subject is specific, but subject lacks the conventional `fix:` type prefix, triggering the 0.3 cap.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 16665,
          "output_tokens": 693,
          "cached_tokens": 14362,
          "wall_ms": 15629
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.3,
          "sd": 0,
          "judge_sd_mean": 0,
          "variance_ratio": null,
          "variance_ratio_unavailable": "judge_sd_zero",
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "685b393dc10f191d27b4cc3fa60d586d58f457365bbd6659ea26b795c6aa3aca",
              "status": "measured",
              "samples": [
                0.3,
                0.3,
                0.3
              ],
              "judge_sample_hashes": [
                "771bf984eee4ac6f8d240866e469fafb02a07dd2e8f5bcaa1e223c07e026b1f2",
                "9327e2afda40b307f048b404c31fac14ac7223d460f8fa063020f4fbb2fb07d8",
                "98be4592ed690e267a7cb6c20726ea4ba4cc2ab89b48b508e8410aa39fc148b0"
              ],
              "mean": 0.3,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 3663,
                "output_tokens": 318,
                "cached_tokens": 2145,
                "wall_ms": 6028
              },
              "judge_usage": {
                "input_tokens": 16557,
                "output_tokens": 655,
                "cached_tokens": 14290,
                "wall_ms": 14959
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "99d6eb26c92cbe2b478f9980c66a9515179e0a8845e46cf64bf941139539be25",
              "status": "measured",
              "samples": [
                0.3,
                0.3,
                0.3
              ],
              "judge_sample_hashes": [
                "24aa77eebc23c28f452e44a6f7266af2ad213a225f8eb73681860418f383f288",
                "9c0bb56119af069ddaee55fa6dcd662e7ba62814541b70a2d406477672897cab",
                "56bbc549c6b5b528c60e36a2b33971c7d4379870c0a20d6d3e66f0650d97988f"
              ],
              "mean": 0.3,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 3663,
                "output_tokens": 335,
                "cached_tokens": 3661,
                "wall_ms": 6389
              },
              "judge_usage": {
                "input_tokens": 16569,
                "output_tokens": 777,
                "cached_tokens": 14298,
                "wall_ms": 16335
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "4f8747d1b186626d570c87ea223d75f840f2186675fd13a0127b90abfab53615",
              "status": "measured",
              "samples": [
                0.3,
                0.3,
                0.3
              ],
              "judge_sample_hashes": [
                "97205b31d50757057faa21bd165e0524ae703237ac8f896870f616af9bda0617",
                "07f025bdad47f5fed4ead9cb780feab3954a1da8f8616d4a429ac3ea85f3a38d",
                "f046b5fbf5c648305acca07a91418828e357bdfbfcfd78358e84ed93ba272380"
              ],
              "mean": 0.3,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 3663,
                "output_tokens": 370,
                "cached_tokens": 3661,
                "wall_ms": 6810
              },
              "judge_usage": {
                "input_tokens": 16665,
                "output_tokens": 693,
                "cached_tokens": 14362,
                "wall_ms": 15629
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.837778,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.3,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.837778,
    "baseline_score": 0.3,
    "delta": 0.537778,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "ca69c3436d0dbfa5db3258fd227f7d84eca8f3737a41ac09e17c35448dd0bf4b",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 8667,
      "mean_output_tokens": 292.33,
      "mean_cost_usd_per_call": 0.040515,
      "median_wall_ms": 6110,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 3663,
      "mean_output_tokens": 341,
      "mean_cost_usd_per_call": 0.021472,
      "median_wall_ms": 6389,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.019043,
    "skill_incremental_cost_usd_per_1k_calls": 19.043,
    "output_tokens_delta": -48.67,
    "median_wall_ms_delta": -279,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.20812,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}