{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-sonnet-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T04:45:51.283Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T04:45:51.283Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.863333,
        "mean": 0.863333,
        "stddev": 0,
        "samples": [
          0.87,
          0.87,
          0.85
        ],
        "generation_hash": "a526401889af7e42893d07e3f12a89b4a866bb9a9acecc67cc92464d0ab95117",
        "judge_sample_hashes": [
          "cc9271aba27fcb121b269af94994720090960d2fd71e3ab06e2b32f141e74035",
          "0a088c03a72ea3beedfaef7c9c1afdde2400ce3b673feb9bab40f1b58528250c",
          "e577afb9d8f7052258f5c0171f24fd5e8876cd5391c219bf09c09619b4afcfa9"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix:` prefix with a specific subject and a body explaining the wrong-unit cause, user-facing impact, and handling of already-issued links.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43872,
          "output_tokens": 716,
          "cached_tokens": 32499,
          "wall_ms": 27104
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.863333,
          "sd": 0,
          "judge_sd_mean": 0.007698,
          "variance_ratio": 0,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "f1e9db9d3bc7f211777934f59d14da26b82dd84b0341b7513ccb1b17a721d55b",
              "status": "measured",
              "samples": [
                0.87,
                0.86,
                0.86
              ],
              "judge_sample_hashes": [
                "13430cbd5e0c16ff97574f0e388f0a4e93092f95d79e2c3b0a86db919af3784e",
                "3d9edd33cd04d6b39d40fc6771539524eea71639891ec2879b1ace07f4ef84dc",
                "08962701e938699df602302ee450a8a0f4cca3aee5a80d94a06c5a3f9226c785"
              ],
              "mean": 0.863333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 17737,
                "output_tokens": 176,
                "cached_tokens": 17735,
                "wall_ms": 3647
              },
              "judge_usage": {
                "input_tokens": 43845,
                "output_tokens": 830,
                "cached_tokens": 32481,
                "wall_ms": 18401
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "0d51da6ce4f9d25871b4d3f2b1ea3ba34b627c4d563c5783a0ef4a1ba2f5b7d3",
              "status": "measured",
              "samples": [
                0.87,
                0.86,
                0.86
              ],
              "judge_sample_hashes": [
                "8f02cf62a1eb0a42e2a233203eace2743c98f9e37c3dd2eb5046ce378f582398",
                "d954ce87fe5a81523bd4be204db8dc8882eb9db0c2b08db693b55a29889473d5",
                "d134e5fe1846fbbec7911fa40978aa6fe04592d75fb51dcc80c92a2bf3e8d628"
              ],
              "mean": 0.863333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 17737,
                "output_tokens": 179,
                "cached_tokens": 17735,
                "wall_ms": 3863
              },
              "judge_usage": {
                "input_tokens": 43854,
                "output_tokens": 774,
                "cached_tokens": 32487,
                "wall_ms": 20830
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "a526401889af7e42893d07e3f12a89b4a866bb9a9acecc67cc92464d0ab95117",
              "status": "measured",
              "samples": [
                0.87,
                0.87,
                0.85
              ],
              "judge_sample_hashes": [
                "cc9271aba27fcb121b269af94994720090960d2fd71e3ab06e2b32f141e74035",
                "0a088c03a72ea3beedfaef7c9c1afdde2400ce3b673feb9bab40f1b58528250c",
                "e577afb9d8f7052258f5c0171f24fd5e8876cd5391c219bf09c09619b4afcfa9"
              ],
              "mean": 0.863333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 17737,
                "output_tokens": 185,
                "cached_tokens": 17735,
                "wall_ms": 5322
              },
              "judge_usage": {
                "input_tokens": 43872,
                "output_tokens": 716,
                "cached_tokens": 32499,
                "wall_ms": 27104
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.85,
        "mean": 0.85,
        "stddev": 0.006667,
        "samples": [
          0.85,
          0.85,
          0.85
        ],
        "generation_hash": "d441437369f3074be0f218f22e19a10f2536aba40009aa4a98cf3b7c9b3c4b18",
        "judge_sample_hashes": [
          "3d0940fd5a3a4d008963d8191a6889de49b299b2b69df49a95a133bc8a2c8fd6",
          "892e5e53c573702c25722edda2db414098cbf1a231e7ffe33a603302d28fa0f8",
          "6f9f4582c2a90575f7554ea73e1ef0a8194d4f3939ce13def80c1c58c56691f3"
        ],
        "threshold": 0.7,
        "reason": "Uses conventional `fix(auth):` prefix, specific subject, and body conveys both user-facing impact (1h vs 24h expiry) and cause (wrong time unit); exemplary but slightly redundant file-path restatement.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43737,
          "output_tokens": 892,
          "cached_tokens": 32409,
          "wall_ms": 38868
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.85,
          "sd": 0.006667,
          "judge_sd_mean": 0.005774,
          "variance_ratio": 1.154659,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "1b467c137469e59fb5738aa4ff98fb87bbbbf49b7a1f0a24aabed9a499931b33",
              "status": "measured",
              "samples": [
                0.86,
                0.85,
                0.86
              ],
              "judge_sample_hashes": [
                "65119c607ae2c59172b4e62cd784e44da5b9c10e07f74ddd9e8ea260595b875f",
                "bc4906c601a669fcc6a12cad92cb133a08ca2ef99c49d8c9ce98b57f140a32b3",
                "5f25d8fc5b53838fc727d64174c299372307b69b8de1baae93f35e2e219b66d8"
              ],
              "mean": 0.856667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12733,
                "output_tokens": 238,
                "cached_tokens": 12731,
                "wall_ms": 4057
              },
              "judge_usage": {
                "input_tokens": 44031,
                "output_tokens": 991,
                "cached_tokens": 32605,
                "wall_ms": 23070
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "d911dd827c12da60acdbe6092012df99704fe1941b00c4bc6a30f7383e790edd",
              "status": "measured",
              "samples": [
                0.83,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "03f6381384fc64ca1cea4ce0aa999e47b1acd370656b584dc04a89e36d5a33c9",
                "ee2e5b68890327ecffc5e088956643ec17015fa601ab7c8ffc72d2a209b0f197",
                "5a5bbb6c15f3f57241c710074e16e9055c334b1eae0074b822a39ab5c8cb3070"
              ],
              "mean": 0.843333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12733,
                "output_tokens": 137,
                "cached_tokens": 12731,
                "wall_ms": 4616
              },
              "judge_usage": {
                "input_tokens": 43728,
                "output_tokens": 918,
                "cached_tokens": 32403,
                "wall_ms": 20057
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "d441437369f3074be0f218f22e19a10f2536aba40009aa4a98cf3b7c9b3c4b18",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "3d0940fd5a3a4d008963d8191a6889de49b299b2b69df49a95a133bc8a2c8fd6",
                "892e5e53c573702c25722edda2db414098cbf1a231e7ffe33a603302d28fa0f8",
                "6f9f4582c2a90575f7554ea73e1ef0a8194d4f3939ce13def80c1c58c56691f3"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12733,
                "output_tokens": 140,
                "cached_tokens": 12731,
                "wall_ms": 4430
              },
              "judge_usage": {
                "input_tokens": 43737,
                "output_tokens": 892,
                "cached_tokens": 32409,
                "wall_ms": 38868
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.863333,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.85,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.863333,
    "baseline_score": 0.85,
    "delta": 0.013333,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "9750ab119e3c09922f84b4188d1040d0cd524223a12e92841023b139fb358517",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 17737,
      "mean_output_tokens": 180,
      "mean_cost_usd_per_call": 0.037274,
      "median_wall_ms": 3863,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 12733,
      "mean_output_tokens": 171.67,
      "mean_cost_usd_per_call": 0.027183,
      "median_wall_ms": 4430,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.010091,
    "skill_incremental_cost_usd_per_1k_calls": 10.091,
    "output_tokens_delta": 8.33,
    "median_wall_ms_delta": -567,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.478245,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}