{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-opus-5",
    "model_release_date": "2026-07-24",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-23T07:08:39.029Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5",
      "reported_models": [
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-23T07:08:39.029Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.803809,
        "mean": 0.803809,
        "stddev": 0.103431,
        "samples": [
          0.87,
          0.87,
          0.87
        ],
        "generation_hash": "4ae269d33831032647e00faeb2d308028768f4181167c24ed6d659842c1dcfe2",
        "judge_sample_hashes": [
          "695f8793668c07455e386fbe3aa04d395e96b6cc1982e0b5aa17a3be9bf96a2f",
          "3f29e0c74cc3d19973cce6b29510997769e063708a229dccfce0071d6f4514ea",
          "44839903b377f8dbc7d7130e03b415d3f3c1d80112aa0ee720c87050ccd709dc"
        ],
        "threshold": 0.7,
        "reason": "Removes all three what-comments (a), the commented-out legacyRate (b), and the bare TODO with justification (c); remaining comments document mutation contract, non-idempotence, and stale-total risk — intent, not what (d).",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 19503,
          "output_tokens": 1223,
          "cached_tokens": 16254,
          "wall_ms": 32032
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 7,
          "n_measured": 7,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.803809,
          "sd": 0.103431,
          "judge_sd_mean": 0.021003,
          "variance_ratio": 4.924582,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "ef528cf258194e94ce99de38322a2e1f5668bdc778f45c05376ef287e2697c24",
              "status": "measured",
              "samples": [
                0.86,
                0.84,
                0.85
              ],
              "judge_sample_hashes": [
                "f3b334e93b27b21ac47a999f89b311fe7d811a2d87a5bf2dd462a5e06e6e4c7a",
                "44911ff98bedd8bb7dbea1bf03675b7af9321c15861502533aa94c729714a6da",
                "c1bcd56b48cc7351f385c3fb84bf443dead7f96c681b362ce61b80abd3fc6b40"
              ],
              "mean": 0.85,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 8187,
                "output_tokens": 1778,
                "cached_tokens": 540,
                "wall_ms": 26147
              },
              "judge_usage": {
                "input_tokens": 19158,
                "output_tokens": 1555,
                "cached_tokens": 16024,
                "wall_ms": 29584
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "742f7ecc47fd7cf15794d117ba1353a5393ce5bbfb84697ae244b60888f0e942",
              "status": "measured",
              "samples": [
                0.6,
                0.55,
                0.6
              ],
              "judge_sample_hashes": [
                "eab1eddb43c8128bb574fb914125f89a7c8141d8891727c50106a81709d7dfc6",
                "d65189c4aed3a158926af8dcd64d6b4399612d8c1020db41593b6821d61be0ec",
                "37e3eaa19e0f3136203ed21cb5ca104c7f88cd8b42a0b97ad54e1bf17accae94"
              ],
              "mean": 0.583333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 8187,
                "output_tokens": 1520,
                "cached_tokens": 8185,
                "wall_ms": 23218
              },
              "judge_usage": {
                "input_tokens": 19359,
                "output_tokens": 1966,
                "cached_tokens": 16158,
                "wall_ms": 30031
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "661199bc80a3dceed98f089253756a54be0ad74a9876f7f0f1af8582772bfadb",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "d74cf9096114f1630e96766bda8ff2c9155cae790949d9e745e6203e543e5a72",
                "446d919bba5b8c9ac4d712504f5e08fd1a7e29ecbd2ec969e563132e6b535841",
                "891e48daf73d9463bd6a6371df8b532f55e0960da8a82bc6cd9bc6f6c4d36526"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 8187,
                "output_tokens": 1672,
                "cached_tokens": 8185,
                "wall_ms": 23322
              },
              "judge_usage": {
                "input_tokens": 19230,
                "output_tokens": 1653,
                "cached_tokens": 16072,
                "wall_ms": 26799
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "cde921d8ac7d8a282ae1ade3ed0cec146e723d4dca343502b0dbfe4b421a858a",
              "status": "measured",
              "samples": [
                0.85,
                0.86,
                0.87
              ],
              "judge_sample_hashes": [
                "efd7cbcff59449232fed9b0937872a161dba7ce01252f892f18b0881595581ef",
                "cabd42db5270659a494d63d351ff224361376f59316e18e8eb545471ed7925e4",
                "e3ab36747ff7ab3576a6c4bb54d2854f07bb69fa3d57dbc88b0f7a0da709c46c"
              ],
              "mean": 0.86,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 8187,
                "output_tokens": 2038,
                "cached_tokens": 8185,
                "wall_ms": 27330
              },
              "judge_usage": {
                "input_tokens": 19533,
                "output_tokens": 1430,
                "cached_tokens": 16274,
                "wall_ms": 25016
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "1a4c5aea8933db28b884ad26aa485460a6241b660f21c92577a4343e01cd1bb5",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "9bd5c91cb6315665f8b0abc1aa9ff9ba91abeb545c18545bd0f021631ec97745",
                "08c0a5932a6168a2b7e1a1b6313ecb449fd4fab9d3edb12fe1ed9d63a25d9fd7",
                "39c1f8910a9cf00823c63988533e94987960587e684ca9095c38db4c71cf5ad9"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 8187,
                "output_tokens": 1789,
                "cached_tokens": 8185,
                "wall_ms": 24360
              },
              "judge_usage": {
                "input_tokens": 18597,
                "output_tokens": 1737,
                "cached_tokens": 15650,
                "wall_ms": 27760
              }
            },
            {
              "draw_index": 5,
              "generation_hash": "d61412a3b394e49f6c8ad13a06609006d69fc526fa909a7503758ad5ce4ba187",
              "status": "measured",
              "samples": [
                0.82,
                0.82,
                0.65
              ],
              "judge_sample_hashes": [
                "d06b5c046a05cd6cd04e57926eee79d852f31588d77d53da48f16fe875247f36",
                "a501ecd608d91585dc3b1386dc49537ec5a12588cfb314d6741ee0d2707b44f9",
                "54ce64ca438f29af99f1a61f64bba2a621a0e6be4a0aef568ed379a958528151"
              ],
              "mean": 0.763333,
              "stddev": 0.09815,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 8187,
                "output_tokens": 1469,
                "cached_tokens": 8185,
                "wall_ms": 24377
              },
              "judge_usage": {
                "input_tokens": 18849,
                "output_tokens": 2709,
                "cached_tokens": 15818,
                "wall_ms": 40730
              }
            },
            {
              "draw_index": 6,
              "generation_hash": "4ae269d33831032647e00faeb2d308028768f4181167c24ed6d659842c1dcfe2",
              "status": "measured",
              "samples": [
                0.87,
                0.87,
                0.87
              ],
              "judge_sample_hashes": [
                "695f8793668c07455e386fbe3aa04d395e96b6cc1982e0b5aa17a3be9bf96a2f",
                "3f29e0c74cc3d19973cce6b29510997769e063708a229dccfce0071d6f4514ea",
                "44839903b377f8dbc7d7130e03b415d3f3c1d80112aa0ee720c87050ccd709dc"
              ],
              "mean": 0.87,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 8187,
                "output_tokens": 1876,
                "cached_tokens": 8185,
                "wall_ms": 27536
              },
              "judge_usage": {
                "input_tokens": 19503,
                "output_tokens": 1223,
                "cached_tokens": 16254,
                "wall_ms": 32032
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.566667,
        "mean": 0.566667,
        "stddev": 0.033334,
        "samples": [
          0.6,
          0.55,
          0.55
        ],
        "generation_hash": "27a7abe8938979818de5f1152730d36a9717f186fdea8f1eec45b1a08e0de2c3",
        "judge_sample_hashes": [
          "ae5798680777bc6c6493fcdd816e71de22c255a9fe8c12d9979abe45e727b926",
          "b485911af0769e091a3f897347add173b3ca9dea1451824211d3f25ac4a451d4",
          "9fcae99ab9134d4e44ce72fd1d18e9d995720b3b383c0ab4dd6ea2f483fe0870"
        ],
        "threshold": 0.7,
        "reason": "Meets (a), (b), and adds a useful intent/mutation doc comment, but leaves an unimplemented TODO and the '10% off' comment restates what 0.9 does rather than why.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 18678,
          "output_tokens": 2371,
          "cached_tokens": 15704,
          "wall_ms": 36642
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.566667,
          "sd": 0.033334,
          "judge_sd_mean": 0.019245,
          "variance_ratio": 1.732086,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "88b0158403bbe83999e24b529a20a13a6405df8810a44723a5568a539f07dbdb",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "17fb75253dd9751d20fdeaa190ec98b20844e7cf83eef3fc9afefe02f1d75316",
                "2a9e9711729c7a56126174d70563eb713b1c7478fdc91afd47dfe842f2d34c0d",
                "9b55671e6d9dbad8e3f1fb038514409cbd180d1cb323f42e175768c7dc073a25"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 4724,
                "output_tokens": 1366,
                "cached_tokens": 3150,
                "wall_ms": 20894
              },
              "judge_usage": {
                "input_tokens": 18669,
                "output_tokens": 1630,
                "cached_tokens": 15698,
                "wall_ms": 26360
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "ad51564ae49a988ef98c442d911a3556c861c1f0fecee8ec9867cc8306f70362",
              "status": "measured",
              "samples": [
                0.55,
                0.5,
                0.55
              ],
              "judge_sample_hashes": [
                "eb539c72e765f4ed2c198b29a7c6bae13e80e77acbd2dc92cd4f090dd5eb29c9",
                "93c35b318783351116065c9dbd7425f1014f2602caf7f42adde673c2534289f2",
                "2c6a53132bff858541511e27912945a0cb3b5805688a52c157df58fb18f68716"
              ],
              "mean": 0.533333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 4724,
                "output_tokens": 999,
                "cached_tokens": 4722,
                "wall_ms": 15021
              },
              "judge_usage": {
                "input_tokens": 18465,
                "output_tokens": 2755,
                "cached_tokens": 15562,
                "wall_ms": 41170
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "27a7abe8938979818de5f1152730d36a9717f186fdea8f1eec45b1a08e0de2c3",
              "status": "measured",
              "samples": [
                0.6,
                0.55,
                0.55
              ],
              "judge_sample_hashes": [
                "ae5798680777bc6c6493fcdd816e71de22c255a9fe8c12d9979abe45e727b926",
                "b485911af0769e091a3f897347add173b3ca9dea1451824211d3f25ac4a451d4",
                "9fcae99ab9134d4e44ce72fd1d18e9d995720b3b383c0ab4dd6ea2f483fe0870"
              ],
              "mean": 0.566667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 4724,
                "output_tokens": 1076,
                "cached_tokens": 4722,
                "wall_ms": 16114
              },
              "judge_usage": {
                "input_tokens": 18678,
                "output_tokens": 2371,
                "cached_tokens": 15704,
                "wall_ms": 36642
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.803809,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.566667,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.803809,
    "baseline_score": 0.566667,
    "delta": 0.237142,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "34107e18c90f3b2087eb88dd465c15982622ca5e21d69c01fa65727b72092a89",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 7,
      "mean_input_tokens": 8187,
      "mean_output_tokens": 1734.57,
      "mean_cost_usd_per_call": 0.084299,
      "median_wall_ms": 24377,
      "wall_ms_p25": 23322,
      "wall_ms_p75": 27330,
      "wall_ms_iqr": 4008
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 4724,
      "mean_output_tokens": 1147,
      "mean_cost_usd_per_call": 0.052295,
      "median_wall_ms": 16114,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.032004,
    "skill_incremental_cost_usd_per_1k_calls": 32.004,
    "output_tokens_delta": 587.57,
    "median_wall_ms_delta": 8263,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.280755,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}