{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-haiku-4-5-20251001",
    "model_release_date": "2025-10-01",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T21:09:35.459Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-4-5-20251001",
      "reported_models": [
        "claude-haiku-4-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T21:09:35.459Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-4-5-20251001": {
          "input_per_mtok": 1,
          "output_per_mtok": 5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "borderline",
        "score": 0.763333,
        "mean": 0.763333,
        "stddev": 0.091591,
        "samples": [
          0.8,
          0.85,
          0.8
        ],
        "generation_hash": "a4590a3288d12682be3aab5252fb0b523985b990f5e638c62d87e3f2c131f11d",
        "judge_sample_hashes": [
          "8b12f6f5785070afe98d524ba2707295aed95383b6d2179ef68efebf05799363",
          "37c92513f6e611f2f6d868ad78d9ace0dad70e86b6e7f01e93354ac42dc274f2",
          "fcc03c8fff1ed015054cb48929dd6a75844efa25aec68923d2df21094c06990c"
        ],
        "threshold": 0.7,
        "reason": "Removes all three restating comments, the commented-out legacy line, and the TODO; added gold-tier comment names the business rule but largely mirrors the code, so not exemplary.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44493,
          "output_tokens": 1018,
          "cached_tokens": 32917,
          "wall_ms": 42573
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 5,
          "n_measured": 5,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.763333,
          "sd": 0.091591,
          "judge_sd_mean": 0.005774,
          "variance_ratio": 15.86266,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "d38a0221d0faaa24284b9308887b0be59ac3a566a0102a15bbd937a7e0ac387d",
              "status": "measured",
              "samples": [
                0.8,
                0.8,
                0.8
              ],
              "judge_sample_hashes": [
                "bdc72f7976770bd5423756e79b46d3ff150260b12e0e1d42bb1cbfbf58291110",
                "27f8874407625028d7cab03f363c7afa456ef91534286759f827230b601940dd",
                "f2133cdba4d41f8b5142312a7340a86c903b2f411c3acfc0ff799d0ed08aff3a"
              ],
              "mean": 0.8,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 16600,
                "output_tokens": 1348,
                "cached_tokens": 0,
                "wall_ms": 15573
              },
              "judge_usage": {
                "input_tokens": 44403,
                "output_tokens": 1072,
                "cached_tokens": 32857,
                "wall_ms": 142065
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "c8103d8622df1525705a949d1a69c0c71d7c6793430a897d6bafe341d9e655a4",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "383d7c07dbd2c53ea0276c1c01180196de341cca664496e40ddda883b6e5515c",
                "c429cc1cbdacf50cc785867e8af2b08fa1de21e6c36732f30355eee486b6a3ab",
                "3e18155fd6a3fcdbda3d0b0b2c280b5d3b2c83ae25edb19e07b2ef731809ce0b"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 16600,
                "output_tokens": 1076,
                "cached_tokens": 16590,
                "wall_ms": 12194
              },
              "judge_usage": {
                "input_tokens": 44361,
                "output_tokens": 1312,
                "cached_tokens": 32829,
                "wall_ms": 27615
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "57ed50ec6bc686f511bcef1ea914c0cc114463e3cc9b583d191d7524a0816dd5",
              "status": "measured",
              "samples": [
                0.8,
                0.8,
                0.8
              ],
              "judge_sample_hashes": [
                "dfa2c8b92ae6d7b09d3e0b9a54668878fe41e2f5a3b2579d1c2251a351359887",
                "5fc3ed96d4629ac04e6e902e5e8428135b3f675912b8fe8bc6489d7d52279697",
                "309103dfe719c5ec47889f5cbf33b06c8cca95215f78dc555fbabb8aff66e47b"
              ],
              "mean": 0.8,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 16600,
                "output_tokens": 1220,
                "cached_tokens": 16590,
                "wall_ms": 13264
              },
              "judge_usage": {
                "input_tokens": 44718,
                "output_tokens": 1036,
                "cached_tokens": 33067,
                "wall_ms": 32796
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "be14a037bfca59ecb04cc6c727a35d8e3a06c3a0e565eb3794ae199638a7d82e",
              "status": "measured",
              "samples": [
                0.8,
                0.8,
                0.8
              ],
              "judge_sample_hashes": [
                "3999a8f16db6d0399246c82f5d8d441743f64f3a35f148386d69cf5c440df5b0",
                "477f321cb24ae2d903d3ae414d3594d3b4a58e80941771fbe2dee30104e89f48",
                "6a32920f24bb0798e4227c4eacc799acab1fef08110aeb90ce96be1ac9592c35"
              ],
              "mean": 0.8,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 16600,
                "output_tokens": 1051,
                "cached_tokens": 16590,
                "wall_ms": 13144
              },
              "judge_usage": {
                "input_tokens": 44610,
                "output_tokens": 893,
                "cached_tokens": 32995,
                "wall_ms": 21024
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "a4590a3288d12682be3aab5252fb0b523985b990f5e638c62d87e3f2c131f11d",
              "status": "measured",
              "samples": [
                0.8,
                0.85,
                0.8
              ],
              "judge_sample_hashes": [
                "8b12f6f5785070afe98d524ba2707295aed95383b6d2179ef68efebf05799363",
                "37c92513f6e611f2f6d868ad78d9ace0dad70e86b6e7f01e93354ac42dc274f2",
                "fcc03c8fff1ed015054cb48929dd6a75844efa25aec68923d2df21094c06990c"
              ],
              "mean": 0.816667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 16600,
                "output_tokens": 1534,
                "cached_tokens": 16590,
                "wall_ms": 16505
              },
              "judge_usage": {
                "input_tokens": 44493,
                "output_tokens": 1018,
                "cached_tokens": 32917,
                "wall_ms": 42573
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.625556,
        "mean": 0.625556,
        "stddev": 0.044264,
        "samples": [
          0.6,
          0.6,
          0.6
        ],
        "generation_hash": "8349ad5efdf1b70e8eb3e79624c3e955dccc0ce99f583f46e45ad3e5d92f74d5",
        "judge_sample_hashes": [
          "04fe48b1ed5a0505a5017c474104d7d5ec6370a845b9fa7e92fd81375bc4088c",
          "a7a2ac490447b93a18cf71591a7a6548d96409dbef5e8353cc45326f5ac319a1",
          "917dc84f73a5f6b58cdfbae7dcdbd90263389e1292641de8785ac54b2834833b"
        ],
        "threshold": 0.7,
        "reason": "Met (a) and (b) by deleting restating comments and commented-out legacy line, but left the bare TODO dangling (c) and added no intent comment (d).",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44454,
          "output_tokens": 1089,
          "cached_tokens": 32891,
          "wall_ms": 90617
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.625556,
          "sd": 0.044264,
          "judge_sd_mean": 0.02457,
          "variance_ratio": 1.801547,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "ceb01b99011835a9ea7976d2bb42b290627aeb67b4ae84691d56ae92e1f9fab1",
              "status": "measured",
              "samples": [
                0.62,
                0.76,
                0.65
              ],
              "judge_sample_hashes": [
                "9806d3ceff0dcbc442221ffe16d975b95b05b09ca4576f6edbfee8dbd8a4563a",
                "b019ea39a9bb506df40c41aafef2153d62f6275ac5a9160a82e86aaeb739dab3",
                "9f18b8b2bd4d46b5753c603f5513230e6050ff72833db52d62c8cc09e38eb99f"
              ],
              "mean": 0.676667,
              "stddev": 0.073711,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14139,
                "output_tokens": 702,
                "cached_tokens": 6184,
                "wall_ms": 9621
              },
              "judge_usage": {
                "input_tokens": 44496,
                "output_tokens": 1580,
                "cached_tokens": 32919,
                "wall_ms": 29997
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "5bf8bdf65f393acd63c0d9a45642333fe530d3e602e108cd3bd9bf0114a5b995",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "bc4c948d633b7cb4e709c1e09ea0699a0e618ff3782023c6826bb2478702121d",
                "d6a59421952156398ac90830522346356b7b2d96c893a2b5fb1b00d33793970f",
                "6ab08648c89802594c0616c6e17dab9b99890dbd36dcc616521f65ac84ee1567"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14139,
                "output_tokens": 1022,
                "cached_tokens": 14129,
                "wall_ms": 12483
              },
              "judge_usage": {
                "input_tokens": 44397,
                "output_tokens": 1196,
                "cached_tokens": 32853,
                "wall_ms": 29504
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "8349ad5efdf1b70e8eb3e79624c3e955dccc0ce99f583f46e45ad3e5d92f74d5",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "04fe48b1ed5a0505a5017c474104d7d5ec6370a845b9fa7e92fd81375bc4088c",
                "a7a2ac490447b93a18cf71591a7a6548d96409dbef5e8353cc45326f5ac319a1",
                "917dc84f73a5f6b58cdfbae7dcdbd90263389e1292641de8785ac54b2834833b"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14139,
                "output_tokens": 1019,
                "cached_tokens": 14129,
                "wall_ms": 11024
              },
              "judge_usage": {
                "input_tokens": 44454,
                "output_tokens": 1089,
                "cached_tokens": 32891,
                "wall_ms": 90617
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.763333,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.625556,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.763333,
    "baseline_score": 0.625556,
    "delta": 0.137777,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "b07465a1c5983497f793fbcbbac84641a2b44a08ccb0a66d0d3acf0dc36efc00",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 5,
      "mean_input_tokens": 16600,
      "mean_output_tokens": 1245.8,
      "mean_cost_usd_per_call": 0.022829,
      "median_wall_ms": 13264,
      "wall_ms_p25": 12669,
      "wall_ms_p75": 16039,
      "wall_ms_iqr": 3370
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 14139,
      "mean_output_tokens": 914.33,
      "mean_cost_usd_per_call": 0.018711,
      "median_wall_ms": 11024,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.004118,
    "skill_incremental_cost_usd_per_1k_calls": 4.118,
    "output_tokens_delta": 331.47,
    "median_wall_ms_delta": 2240,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.49741,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}