{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-sonnet-5",
    "model_release_date": "2026-06-30",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T04:15:21.408Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T04:15:21.408Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.814444,
        "mean": 0.814444,
        "stddev": 0.02835,
        "samples": [
          0.85,
          0.84,
          0.85
        ],
        "generation_hash": "01224c5f15d74067cee1af7f77a54fa0034bf6790a67169c81a790e4b93d70cb",
        "judge_sample_hashes": [
          "e36ceff37a00559496a2f1a711e8099e06bb1a663ac1ea0a049a34679b3d486d",
          "f18e05211d37871da2c2fa33e2b0d99f63f75d6503d93377eb8e59e4d451652e",
          "ee7e218d02d099e6d5b2f65736936d6c4b2b98ae9926bc5c045990a7a07c3d06"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix:` prefix with a specific subject, and the body states both the user-facing consequence (24h links expiring in ~1h) and the unit-conversion cause.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43596,
          "output_tokens": 1011,
          "cached_tokens": 32315,
          "wall_ms": 20405
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.814444,
          "sd": 0.02835,
          "judge_sd_mean": 0.014162,
          "variance_ratio": 2.001836,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "e14f93e122e5465520ea65a9645fcc0a06026f597a15be43daf7f142018be00c",
              "status": "measured",
              "samples": [
                0.78,
                0.83,
                0.8
              ],
              "judge_sample_hashes": [
                "a57754f0ef42706561c7b67c62ffd3480611b173d83b62eac2ee91e9bf48f6c3",
                "1a87828f4e2392823bd8b37e8bd081e7564349dc99a9d22378528af6d8bf540a",
                "b01043be04ab6624365edf1597e0694f935ba01488f99a8b066d0906cf41f799"
              ],
              "mean": 0.803333,
              "stddev": 0.025166,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 24155,
                "output_tokens": 65,
                "cached_tokens": 3406,
                "wall_ms": 3748
              },
              "judge_usage": {
                "input_tokens": 43512,
                "output_tokens": 2036,
                "cached_tokens": 32259,
                "wall_ms": 35724
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "b3cdbf10e1d02d897916d17dc3b2d98917718abe02d23d39c1a675af78b8191b",
              "status": "measured",
              "samples": [
                0.8,
                0.8,
                0.78
              ],
              "judge_sample_hashes": [
                "b65dd38429283f31b4880562a785713fcd3fbbcab1b23086a5c2d14129e454b0",
                "c512c9621d00a4408628dd9c981d04285e000de14e2427c4657b1a458249cd51",
                "cd78dc8491562a08b51a45734ac51ff666465405a2af6e5bcc5e2f3c5ddc88e8"
              ],
              "mean": 0.793333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 24155,
                "output_tokens": 123,
                "cached_tokens": 24153,
                "wall_ms": 4099
              },
              "judge_usage": {
                "input_tokens": 43686,
                "output_tokens": 1619,
                "cached_tokens": 32375,
                "wall_ms": 31348
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "01224c5f15d74067cee1af7f77a54fa0034bf6790a67169c81a790e4b93d70cb",
              "status": "measured",
              "samples": [
                0.85,
                0.84,
                0.85
              ],
              "judge_sample_hashes": [
                "e36ceff37a00559496a2f1a711e8099e06bb1a663ac1ea0a049a34679b3d486d",
                "f18e05211d37871da2c2fa33e2b0d99f63f75d6503d93377eb8e59e4d451652e",
                "ee7e218d02d099e6d5b2f65736936d6c4b2b98ae9926bc5c045990a7a07c3d06"
              ],
              "mean": 0.846667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 24155,
                "output_tokens": 93,
                "cached_tokens": 24153,
                "wall_ms": 3888
              },
              "judge_usage": {
                "input_tokens": 43596,
                "output_tokens": 1011,
                "cached_tokens": 32315,
                "wall_ms": 20405
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.802222,
        "mean": 0.802222,
        "stddev": 0.04299,
        "samples": [
          0.78,
          0.8,
          0.79
        ],
        "generation_hash": "7c2b16061e34706a167cc68315864e87f99f5b2f83fae973b997d03f46c9ce22",
        "judge_sample_hashes": [
          "0ed16bdef19172c747770d581d6e791014a030b3f24ec1c08b848ae5afef9fe2",
          "ae91c1ec470824e1878160b43d21570c1d796c45a4463dcc7d55ff62e1ed7b7e",
          "c8ca925373b1dc1e8b6f0252ab937dbeb09ea719bd568d46089ab138705e2f1f"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix(auth):` prefix with a specific subject and a body giving consequence and cause; minor deduction for the internally inconsistent invented detail \"24 minutes' worth ... (~1 hour)\".",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43734,
          "output_tokens": 1536,
          "cached_tokens": 32407,
          "wall_ms": 27346
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.802222,
          "sd": 0.04299,
          "judge_sd_mean": 0.007182,
          "variance_ratio": 5.985798,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "4db7c2a34c2357805ae701291b557a257f74518a18efb78d1d54503e270c522e",
              "status": "measured",
              "samples": [
                0.76,
                0.76,
                0.78
              ],
              "judge_sample_hashes": [
                "453882da0202e1eb0355739b5ae5773bbd40e2dc0f24848c17d430dfcdb7dc96",
                "69e869e36234e0621b591472aa339dbcc441a4b0e00bc218bac0e5d8f2485638",
                "97736fb54599c11ea37f077fdf731fa08ff710244ca3d797c07630285f723377"
              ],
              "mean": 0.766667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19151,
                "output_tokens": 143,
                "cached_tokens": 8503,
                "wall_ms": 5562
              },
              "judge_usage": {
                "input_tokens": 43746,
                "output_tokens": 1518,
                "cached_tokens": 32415,
                "wall_ms": 28797
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "d8f3b18d5fc82ac58fbee4cc7c91ad465c3b569c556451690fd52fef75e1962a",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "1bae29a7cf00c59c1a7a4f18b92f51b9faeaab7c95c0aa366131433c0b0ab1a9",
                "0f9ff738bac22882275a416656cb080ae85cc8cc5fd404f82cecf1fd0b91767b",
                "b3c07680082fb674625b0eafd11a11c3367dae9e0952681f2924136799b8d984"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19151,
                "output_tokens": 100,
                "cached_tokens": 19149,
                "wall_ms": 3246
              },
              "judge_usage": {
                "input_tokens": 43617,
                "output_tokens": 796,
                "cached_tokens": 32329,
                "wall_ms": 17571
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "7c2b16061e34706a167cc68315864e87f99f5b2f83fae973b997d03f46c9ce22",
              "status": "measured",
              "samples": [
                0.78,
                0.8,
                0.79
              ],
              "judge_sample_hashes": [
                "0ed16bdef19172c747770d581d6e791014a030b3f24ec1c08b848ae5afef9fe2",
                "ae91c1ec470824e1878160b43d21570c1d796c45a4463dcc7d55ff62e1ed7b7e",
                "c8ca925373b1dc1e8b6f0252ab937dbeb09ea719bd568d46089ab138705e2f1f"
              ],
              "mean": 0.79,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19151,
                "output_tokens": 139,
                "cached_tokens": 19149,
                "wall_ms": 4159
              },
              "judge_usage": {
                "input_tokens": 43734,
                "output_tokens": 1536,
                "cached_tokens": 32407,
                "wall_ms": 27346
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.814444,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.802222,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.814444,
    "baseline_score": 0.802222,
    "delta": 0.012222,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "14902c6b274a124b3468de16373f890bcefbdba7e878841877914dce07dd4ae3",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 24155,
      "mean_output_tokens": 93.67,
      "mean_cost_usd_per_call": 0.049247,
      "median_wall_ms": 3888,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 19151,
      "mean_output_tokens": 127.33,
      "mean_cost_usd_per_call": 0.039575,
      "median_wall_ms": 4159,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.009672,
    "skill_incremental_cost_usd_per_1k_calls": 9.672,
    "output_tokens_delta": -33.66,
    "median_wall_ms_delta": -271,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.500325,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}