{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-sonnet-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T05:45:44.389Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T05:45:44.389Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.863333,
        "mean": 0.863333,
        "stddev": 0.008819,
        "samples": [
          0.87,
          0.87,
          0.87
        ],
        "generation_hash": "a5abbf4788d5675bb11a3428bae4fa1805eea900106cfeafc679ab205b70d6fc",
        "judge_sample_hashes": [
          "09516a84e63f13f428fe1fe5043ff43e274040f9d5b96e65f27ab5ef28fb196e",
          "cc6ccea67cf06530514486d10e444ba1cbecbc8a3dcca459cc901cb9d71c33d3",
          "44b14b549ff556383dac9b6a5f466e399c96fdbf7fd8f4d652a0f26669f4ca05"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix:` prefix with a specific subject; body conveys user-facing symptom (1h vs 24h), the unit-conversion cause, and impact on already-issued links, slightly verbose.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43860,
          "output_tokens": 765,
          "cached_tokens": 32491,
          "wall_ms": 18212
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.863333,
          "sd": 0.008819,
          "judge_sd_mean": 0.003849,
          "variance_ratio": 2.291244,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "cb287c4de6783e14af8993a77369e11c3c2a3f78cc3fee881282f4c6447150d1",
              "status": "measured",
              "samples": [
                0.86,
                0.87,
                0.87
              ],
              "judge_sample_hashes": [
                "d3835a2a06efd704dd33916ab960559a666975d15a72f6f367c9c3da1f85b615",
                "262c531aa5fb785567d9626cf112c60cdd7b7dff06e896c3ffa1d5d9d9458fb9",
                "25fca25819b7f7a56c3c8dc56b1556b7be81af3f19e2a630403427906fe6d769"
              ],
              "mean": 0.866667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 17737,
                "output_tokens": 167,
                "cached_tokens": 17735,
                "wall_ms": 4013
              },
              "judge_usage": {
                "input_tokens": 43818,
                "output_tokens": 761,
                "cached_tokens": 32463,
                "wall_ms": 17364
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "0f00a2d8d1dbb49213d31c1e7b67c2c3eeac9a6d427297b501341016ab0d7729",
              "status": "measured",
              "samples": [
                0.85,
                0.86,
                0.85
              ],
              "judge_sample_hashes": [
                "52544b2bd7e488dfe7e13a5fc52932da4aa5190786fcb3769a5e1fb3a176d4a8",
                "a27cb249ceeecc8f393d5a361c40ac69ccee8bbaf7f3c570759f4d9002ea3046",
                "5509c2561d6267477b73c26ff212db7d7200f4ab2ef920d0edd2e1ddabfa5ff4"
              ],
              "mean": 0.853333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 17737,
                "output_tokens": 249,
                "cached_tokens": 17735,
                "wall_ms": 8390
              },
              "judge_usage": {
                "input_tokens": 44064,
                "output_tokens": 1152,
                "cached_tokens": 32627,
                "wall_ms": 32210
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "a5abbf4788d5675bb11a3428bae4fa1805eea900106cfeafc679ab205b70d6fc",
              "status": "measured",
              "samples": [
                0.87,
                0.87,
                0.87
              ],
              "judge_sample_hashes": [
                "09516a84e63f13f428fe1fe5043ff43e274040f9d5b96e65f27ab5ef28fb196e",
                "cc6ccea67cf06530514486d10e444ba1cbecbc8a3dcca459cc901cb9d71c33d3",
                "44b14b549ff556383dac9b6a5f466e399c96fdbf7fd8f4d652a0f26669f4ca05"
              ],
              "mean": 0.87,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 17737,
                "output_tokens": 181,
                "cached_tokens": 17735,
                "wall_ms": 3891
              },
              "judge_usage": {
                "input_tokens": 43860,
                "output_tokens": 765,
                "cached_tokens": 32491,
                "wall_ms": 18212
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.856667,
        "mean": 0.856667,
        "stddev": 0.008819,
        "samples": [
          0.86,
          0.85,
          0.85
        ],
        "generation_hash": "020f467737049d2b7a7ae0af84300f344e73c764864a6d8c2a359db3ad6deecf",
        "judge_sample_hashes": [
          "c7289d1aea788c7a01866a96112e627638a9c10df22f1ce175a5a6564cd346f8",
          "cfbe16ef650e5eb6ac1e5778083dd6256ab7a132c71f5cf069d340d9906b0bce",
          "0f2e9565276b2826ee57e4b95e742dbed09790856118110c6c670baa15c8f668"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix(auth):` prefix with a specific subject and a body conveying both user-facing impact (1h vs 24h) and cause (missing unit conversion); slightly marred by a hedged, conditional final paragraph.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 44001,
          "output_tokens": 977,
          "cached_tokens": 32585,
          "wall_ms": 18566
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.856667,
          "sd": 0.008819,
          "judge_sd_mean": 0.003849,
          "variance_ratio": 2.291244,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "062a44b980ec8f949d26d073eba2b4b3b08e8c30f6cdde6cd2329578807888d4",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "2c189f418aac7993a25e2901da15c60afd45594b6a3e674004590e49d68b15c2",
                "355a95fa29e501c158605e22be3dfe97218f28e671aa68f33ad3b7d8c89fa00a",
                "17aca52a34b88ed5c55e101c6023d8efd10b51215ccb0a634fec38bbc8532f6f"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12733,
                "output_tokens": 263,
                "cached_tokens": 12731,
                "wall_ms": 4139
              },
              "judge_usage": {
                "input_tokens": 44106,
                "output_tokens": 1079,
                "cached_tokens": 32655,
                "wall_ms": 26183
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "9f6449bcc810c4f82a164315fa97565987fc04c5e4cf6ce1575db50e4756b0d0",
              "status": "measured",
              "samples": [
                0.87,
                0.87,
                0.86
              ],
              "judge_sample_hashes": [
                "58dbe8de8baaf8e0b94d9a54abb432a6fc0211a1fda61c8bcce2947fde3b35db",
                "23643eeae46a21f8aec0bf2c84c6eb54d218ee1584d61e05068ad98f1c8adc1e",
                "1e65a4f5dcf6f8b3b67cca46b487e4686aa50e60f797554d0f583c000cd5b4e8"
              ],
              "mean": 0.866667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12733,
                "output_tokens": 191,
                "cached_tokens": 12731,
                "wall_ms": 3919
              },
              "judge_usage": {
                "input_tokens": 43890,
                "output_tokens": 985,
                "cached_tokens": 32511,
                "wall_ms": 28354
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "020f467737049d2b7a7ae0af84300f344e73c764864a6d8c2a359db3ad6deecf",
              "status": "measured",
              "samples": [
                0.86,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "c7289d1aea788c7a01866a96112e627638a9c10df22f1ce175a5a6564cd346f8",
                "cfbe16ef650e5eb6ac1e5778083dd6256ab7a132c71f5cf069d340d9906b0bce",
                "0f2e9565276b2826ee57e4b95e742dbed09790856118110c6c670baa15c8f668"
              ],
              "mean": 0.853333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12733,
                "output_tokens": 228,
                "cached_tokens": 12731,
                "wall_ms": 3943
              },
              "judge_usage": {
                "input_tokens": 44001,
                "output_tokens": 977,
                "cached_tokens": 32585,
                "wall_ms": 18566
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.863333,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.856667,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.863333,
    "baseline_score": 0.856667,
    "delta": 0.006666,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "49f1b9adaf1667e77b1f850a1273364243e05211f1d1b989096902854d563ba3",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 17737,
      "mean_output_tokens": 199,
      "mean_cost_usd_per_call": 0.037464,
      "median_wall_ms": 4013,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 12733,
      "mean_output_tokens": 227.33,
      "mean_cost_usd_per_call": 0.027739,
      "median_wall_ms": 3943,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.009725,
    "skill_incremental_cost_usd_per_1k_calls": 9.725,
    "output_tokens_delta": -28.33,
    "median_wall_ms_delta": 70,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.482855,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}