{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "git-workflow-and-versioning",
    "version": "0.0.0",
    "content_hash": "91c8c72654ee6ef163e49df26d1810cb7a6a91a4d8dee5b9f1bc03fd1fb8ec3a",
    "tokens": 3466
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "4e74150753d06b166d642365405edf382dc1072dc529cfdc1ef552d9b00c7880",
    "case_count": 1,
    "canary": "001f3e62-7789-6bf5-ccfc-7f0c9ab434ae"
  },
  "run": {
    "model_id": "claude-haiku-4-5-20251001",
    "model_release_date": "2025-10-01",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T21:00:55.254Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-4-5-20251001",
      "reported_models": [
        "claude-haiku-4-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T21:00:55.254Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-4-5-20251001": {
          "input_per_mtok": 1,
          "output_per_mtok": 5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "commit-message-conventional-type",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.832222,
        "mean": 0.832222,
        "stddev": 0.001924,
        "samples": [
          0.83,
          0.83,
          0.83
        ],
        "generation_hash": "86629059ec646b0625c4c94986d3c73ff139aa595287cac8de77616276dd12ec",
        "judge_sample_hashes": [
          "0e3733f9971b5c11a6c50faea8959b99dc640ce56ce3a8786cdb9392f42b2570",
          "806c3e3c13dd0b7a05a5bdad998b6357ab7f3a7a7ec8502dc0d5d78b25f354da",
          "744ddc85cb105f71bfb9f75617512203f2e91ecb114efa8a058e29264ed6b50f"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix:` prefix with a specific subject and a body conveying both the user-facing symptom (1h vs 24h expiry) and the wrong-unit cause; idiomatic, slightly redundant closing line.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43647,
          "output_tokens": 805,
          "cached_tokens": 32353,
          "wall_ms": 20795
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.832222,
          "sd": 0.001924,
          "judge_sd_mean": 0.003849,
          "variance_ratio": 0.49987,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "5c93d133ff5791ec6564238e8a8e169d1bcc084683f54262bd63d855cdf7ab76",
              "status": "measured",
              "samples": [
                0.83,
                0.83,
                0.84
              ],
              "judge_sample_hashes": [
                "798b89087db93459b14ae06df11835bb2bbbfd2d7bc6debd473b34c18b165abd",
                "e77d927136ce6b79e36502d3dac4c05e74f8a3c3b629fa98fc47407d351d927f",
                "12b2ed7960e91ad5e386cf67c2434a103ea46de2cc2ec3217cbda41c7b3ed0ff"
              ],
              "mean": 0.833333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 17698,
                "output_tokens": 437,
                "cached_tokens": 0,
                "wall_ms": 6290
              },
              "judge_usage": {
                "input_tokens": 43671,
                "output_tokens": 819,
                "cached_tokens": 32369,
                "wall_ms": 21021
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "07e1171c7993dd38b112413b7fc516e8df8a58b08154d19c4924a7056d80c751",
              "status": "measured",
              "samples": [
                0.83,
                0.83,
                0.84
              ],
              "judge_sample_hashes": [
                "91e7ea3bbac00af4746b570f3ffeecbe77bea74ec8dc26e534c032d474cecb31",
                "3022c94f1046f5c198deafee62782794a6a0f66a3eb15a834f1a3a3399870b3b",
                "4d0c533bb37bc69e2f4aff9015a0d687d6efdb1245fe0e73e035daf521afbb85"
              ],
              "mean": 0.833333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 17698,
                "output_tokens": 727,
                "cached_tokens": 17688,
                "wall_ms": 9338
              },
              "judge_usage": {
                "input_tokens": 43644,
                "output_tokens": 861,
                "cached_tokens": 32351,
                "wall_ms": 30837
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "86629059ec646b0625c4c94986d3c73ff139aa595287cac8de77616276dd12ec",
              "status": "measured",
              "samples": [
                0.83,
                0.83,
                0.83
              ],
              "judge_sample_hashes": [
                "0e3733f9971b5c11a6c50faea8959b99dc640ce56ce3a8786cdb9392f42b2570",
                "806c3e3c13dd0b7a05a5bdad998b6357ab7f3a7a7ec8502dc0d5d78b25f354da",
                "744ddc85cb105f71bfb9f75617512203f2e91ecb114efa8a058e29264ed6b50f"
              ],
              "mean": 0.83,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 17698,
                "output_tokens": 374,
                "cached_tokens": 17688,
                "wall_ms": 6109
              },
              "judge_usage": {
                "input_tokens": 43647,
                "output_tokens": 805,
                "cached_tokens": 32353,
                "wall_ms": 20795
              }
            }
          ]
        }
      },
      {
        "id": "commit-message-conventional-type",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.834445,
        "mean": 0.834445,
        "stddev": 0.015753,
        "samples": [
          0.83,
          0.84,
          0.85
        ],
        "generation_hash": "00c6dfb7fefbc143c708d52603864a5601e947fccffbf9b1a951c5285243566e",
        "judge_sample_hashes": [
          "655edf70a75ed4c34e1b678063c7470a04980974338324d242cb611d67509bdf",
          "7e03da5f01a3caf9e8ddb1bd2ad4100c30efd6f87ff54f34585c75eab6b6036c",
          "9ab008a7fb13c889626a93b35c00648dfc8369975f2a3c446bd61fd553b8ba12"
        ],
        "threshold": 0.7,
        "reason": "Uses `fix(auth):` prefix with specific subject and a body conveying both user-facing impact (~1h vs 24h) and cause (wrong time unit); subject is a noun phrase, not imperative.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "c17f2f214cec220ac31af3cfe88a52393c254cea8f9a94fc16e668f668fa88c3"
        },
        "judge_usage": {
          "input_tokens": 43581,
          "output_tokens": 1001,
          "cached_tokens": 32309,
          "wall_ms": 32668
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.834445,
          "sd": 0.015753,
          "judge_sd_mean": 0.01035,
          "variance_ratio": 1.522029,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "0bec6819625e220bdecf8c04ee25642fa50646846ab9c99fa357f4f3cd8280eb",
              "status": "measured",
              "samples": [
                0.84,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "cd6a9dba97ceccd7efa92142a3661135c3abe3494c9365dac560a1b563173d1d",
                "6e0b76cb8e0a2b61cd57a2881a89901b52450cf5cd2261fc4f31086b475081b0",
                "5224327966ed0e0e01edbe13bb603873638b67e8bd034c2b4fd222c5631251a6"
              ],
              "mean": 0.846667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14085,
                "output_tokens": 451,
                "cached_tokens": 6184,
                "wall_ms": 6700
              },
              "judge_usage": {
                "input_tokens": 43647,
                "output_tokens": 885,
                "cached_tokens": 32353,
                "wall_ms": 18484
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "878733900005f14b28bc1e2d28c39c98cd6f8375b4ac6a305d887916634804ba",
              "status": "measured",
              "samples": [
                0.82,
                0.83,
                0.8
              ],
              "judge_sample_hashes": [
                "f0d1c394c7d88922d66fea5f29febdca96e168e0cd9c3f59196d4484ca25f107",
                "69e2a9baec8f6ac86c917481f2f6306afacecf3ed667b3d624a70a69926ff941",
                "8585dd9410953bd20fb33a3426ac0976d43e540f8d6f802035894b4ebe7ac916"
              ],
              "mean": 0.816667,
              "stddev": 0.015275,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14085,
                "output_tokens": 908,
                "cached_tokens": 14075,
                "wall_ms": 12243
              },
              "judge_usage": {
                "input_tokens": 43656,
                "output_tokens": 1044,
                "cached_tokens": 32359,
                "wall_ms": 26545
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "00c6dfb7fefbc143c708d52603864a5601e947fccffbf9b1a951c5285243566e",
              "status": "measured",
              "samples": [
                0.83,
                0.84,
                0.85
              ],
              "judge_sample_hashes": [
                "655edf70a75ed4c34e1b678063c7470a04980974338324d242cb611d67509bdf",
                "7e03da5f01a3caf9e8ddb1bd2ad4100c30efd6f87ff54f34585c75eab6b6036c",
                "9ab008a7fb13c889626a93b35c00648dfc8369975f2a3c446bd61fd553b8ba12"
              ],
              "mean": 0.84,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14085,
                "output_tokens": 950,
                "cached_tokens": 14075,
                "wall_ms": 11731
              },
              "judge_usage": {
                "input_tokens": 43581,
                "output_tokens": 1001,
                "cached_tokens": 32309,
                "wall_ms": 32668
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.832222,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.834445,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.832222,
    "baseline_score": 0.834445,
    "delta": -0.002223,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "9d2667c7cc74ff3cb7c74eea84f2b69a76a31cb0f006b34464385bdffd87c667",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 17698,
      "mean_output_tokens": 512.67,
      "mean_cost_usd_per_call": 0.020261,
      "median_wall_ms": 6290,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 14085,
      "mean_output_tokens": 769.67,
      "mean_cost_usd_per_call": 0.017933,
      "median_wall_ms": 11731,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.002328,
    "skill_incremental_cost_usd_per_1k_calls": 2.328,
    "output_tokens_delta": -257,
    "median_wall_ms_delta": -5441,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.48129,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}