{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-opus-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T06:17:46.758Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-opus-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T06:17:46.758Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5-5": {
          "input_per_mtok": 4,
          "output_per_mtok": 20,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.855556,
        "mean": 0.855556,
        "stddev": 0.009623,
        "samples": [
          0.85,
          0.85,
          0.85
        ],
        "generation_hash": "3215d9323a27332936ce583adc421367146b4fc96326540d2adb050ef2bda4bb",
        "judge_sample_hashes": [
          "f94165f235488a188ec9f3fc4f1379e6ddf4f80ed191abcf1fd2c70a56e70f9c",
          "58baf33aacf856ae0ea8c70b59d84c928e2da82a600572271ecde0bfc7592501",
          "7d29b868bd09a3e717f6f4ec3fe055d0f383244e5dabc007e2c3e1e951d52574"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix:` prefix with a specific subject, and the body states the user-facing symptom (links dying in ~1 hour) plus the unit-conversion cause and impact, exceeding a bare diff restatement.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43743,
          "output_tokens": 865,
          "cached_tokens": 32413,
          "wall_ms": 19906
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.855556,
          "sd": 0.009623,
          "judge_sd_mean": 0.001925,
          "variance_ratio": 4.998961,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "5e22549acba2e667000c60a6d7e4f5885f60f3520abfe851852409233bbfd987",
              "status": "measured",
              "samples": [
                0.87,
                0.87,
                0.86
              ],
              "judge_sample_hashes": [
                "bf0885045cbef84576cb4e5fee7a578cb8e1a413a791660eb6d8d7485bf99604",
                "3a081ebb1ffb0774e35bffe5166de5c3ebb5a631eaf606c0288ef515ff90d3aa",
                "b57a5efccfa6bd53f0c5b721a5dffee699e00c214d4685086a1aecb31819e916"
              ],
              "mean": 0.866667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 17733,
                "output_tokens": 438,
                "cached_tokens": 540,
                "wall_ms": 7409
              },
              "judge_usage": {
                "input_tokens": 44256,
                "output_tokens": 933,
                "cached_tokens": 32755,
                "wall_ms": 19365
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "a085bf3a7731020d04ce5849fccc1fa4331117f56ea1d82fac55da7c936b38af",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "75db8dd448ca59be5cac847d3f1ece45becee7993852562f9959fb3fdd7b9bd8",
                "1675eb57a4cd1a4ffa62b7043d284a117893706a3a406139dd77fb6bb6ba9641",
                "625fc90a7e54fb371e43bc0f81e7dc203597cdb9067eaba228ae2e29c642389b"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 17733,
                "output_tokens": 369,
                "cached_tokens": 17731,
                "wall_ms": 6373
              },
              "judge_usage": {
                "input_tokens": 44076,
                "output_tokens": 1132,
                "cached_tokens": 32635,
                "wall_ms": 23828
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "3215d9323a27332936ce583adc421367146b4fc96326540d2adb050ef2bda4bb",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "f94165f235488a188ec9f3fc4f1379e6ddf4f80ed191abcf1fd2c70a56e70f9c",
                "58baf33aacf856ae0ea8c70b59d84c928e2da82a600572271ecde0bfc7592501",
                "7d29b868bd09a3e717f6f4ec3fe055d0f383244e5dabc007e2c3e1e951d52574"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 17733,
                "output_tokens": 219,
                "cached_tokens": 17731,
                "wall_ms": 5345
              },
              "judge_usage": {
                "input_tokens": 43743,
                "output_tokens": 865,
                "cached_tokens": 32413,
                "wall_ms": 19906
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "borderline",
        "score": 0.579167,
        "mean": 0.579167,
        "stddev": 0.322426,
        "samples": [
          0.3,
          0.3,
          0.3
        ],
        "generation_hash": "c2589366031a565db336bafbb4f0148b289b9fd230a823c84be594a0abc48254",
        "judge_sample_hashes": [
          "b1e61b6a8ec484c8a9b2537f18bc0272f6723903f4df2f08505e27de45393874",
          "633d0d4c8d322a23a6205de3b15a876b9552f41ac6f36acea038b9433fb148c1",
          "a071d0cffc6b197bf5abea3af144bc94cc350de1e9caf37c9c1562fdcb72a5db"
        ],
        "threshold": 0.7,
        "reason": "Body explains cause and user-facing impact well and subject is specific, but subject uses 'Fix ...' rather than the conventional `fix:` prefix, triggering the 0.3 cap.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43794,
          "output_tokens": 632,
          "cached_tokens": 32447,
          "wall_ms": 19016
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.579167,
          "sd": 0.322426,
          "judge_sd_mean": 0.001444,
          "variance_ratio": 223.286704,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "3b2043205e98ab266b046f8d9e568315c0f46bd0178caac4e068825c9f6a3b22",
              "status": "measured",
              "samples": [
                0.87,
                0.87,
                0.86
              ],
              "judge_sample_hashes": [
                "1c12d2696b3376b5e224256018b02a96b435233c87a36b197a5c93b1f1939b0c",
                "5a962802f2a24793b788f4e94858ab55ed7d5ae1f94e2d24b57a6e6c200a47cf",
                "1e7e92d06f734f562efbeab31f1687cfb5a1a92ebba6cade9282091836a75127"
              ],
              "mean": 0.866667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12729,
                "output_tokens": 272,
                "cached_tokens": 2144,
                "wall_ms": 6067
              },
              "judge_usage": {
                "input_tokens": 43791,
                "output_tokens": 889,
                "cached_tokens": 32445,
                "wall_ms": 28365
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "0af62f6b5cf5ac4960e17f03107011da1fdeec5cba4a66827da2fea029f0a052",
              "status": "measured",
              "samples": [
                0.3,
                0.3,
                0.3
              ],
              "judge_sample_hashes": [
                "520fb861783867ae643b76cf11c83cb46207a50c09957de755284dfe698033c1",
                "634dec31be007797a6d3603e2f87a54ab687996ab453f9585b75ba95c09e0fe6",
                "b89945f92c67e610efd4361d45f4d089091a8f07a231affa3d48d0c46c7c77af"
              ],
              "mean": 0.3,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12729,
                "output_tokens": 282,
                "cached_tokens": 12727,
                "wall_ms": 5727
              },
              "judge_usage": {
                "input_tokens": 43719,
                "output_tokens": 582,
                "cached_tokens": 32397,
                "wall_ms": 18253
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "44aebf51173c76c51ba643b8c922379c5ac027cafb0eee98ca982b8877e5a877",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "32cb2b6fb401bfdf1bbfacd7023a412a47d6e48c883b163d5e9e5bb077bd1947",
                "c4d5ef9494e2e0a248c9e69d1cd535ebdab8b2406f40fb208c9778661a1fb4af",
                "ef317918953455a06cbf48658984e72572c9e527e448177cd5689621e41299e4"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12729,
                "output_tokens": 395,
                "cached_tokens": 12727,
                "wall_ms": 6586
              },
              "judge_usage": {
                "input_tokens": 44007,
                "output_tokens": 1116,
                "cached_tokens": 32589,
                "wall_ms": 22114
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "c2589366031a565db336bafbb4f0148b289b9fd230a823c84be594a0abc48254",
              "status": "measured",
              "samples": [
                0.3,
                0.3,
                0.3
              ],
              "judge_sample_hashes": [
                "b1e61b6a8ec484c8a9b2537f18bc0272f6723903f4df2f08505e27de45393874",
                "633d0d4c8d322a23a6205de3b15a876b9552f41ac6f36acea038b9433fb148c1",
                "a071d0cffc6b197bf5abea3af144bc94cc350de1e9caf37c9c1562fdcb72a5db"
              ],
              "mean": 0.3,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12729,
                "output_tokens": 373,
                "cached_tokens": 12727,
                "wall_ms": 6933
              },
              "judge_usage": {
                "input_tokens": 43794,
                "output_tokens": 632,
                "cached_tokens": 32447,
                "wall_ms": 19016
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.855556,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.579167,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.855556,
    "baseline_score": 0.579167,
    "delta": 0.276389,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "ee303cee2a58c94511775f4a5432e7179f44877d60cd9ba538c910fb2d6e768a",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 17733,
      "mean_output_tokens": 342,
      "mean_cost_usd_per_call": 0.077772,
      "median_wall_ms": 6373,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 4,
      "mean_input_tokens": 12729,
      "mean_output_tokens": 330.5,
      "mean_cost_usd_per_call": 0.057526,
      "median_wall_ms": 6326.5,
      "wall_ms_p25": 5897,
      "wall_ms_p75": 6759.5,
      "wall_ms_iqr": 862.5
    },
    "skill_incremental_cost_usd_per_call": 0.020246,
    "skill_incremental_cost_usd_per_1k_calls": 20.246,
    "output_tokens_delta": 11.5,
    "median_wall_ms_delta": 46.5,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.47511,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}