{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-opus-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T06:24:35.613Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-opus-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T06:24:35.613Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5-5": {
          "input_per_mtok": 4,
          "output_per_mtok": 20,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.84,
        "mean": 0.84,
        "stddev": 0.020276,
        "samples": [
          0.86,
          0.85,
          0.85
        ],
        "generation_hash": "43bf2a71ada321bc7384bb161dcad24d62beb0cc5293008be6b31ad9dcacbebe",
        "judge_sample_hashes": [
          "b68ec3b27be90469cfeb3c5146eece5a41ff28233d039a4dc56461a7c5c1dc67",
          "56b2aa39842058989a00501fb038a37d44577edc2cbfc7db8e7f058c88edc182",
          "c2b92cad5d4d2673541594d44e37792b7fda19f6ce265542c0b209065d83867b"
        ],
        "threshold": 0.7,
        "reason": "Removes all three restating comments, the commented-out legacyRate, and the bare TODO; remaining JSDoc and inline comments explain intent, mutation, and non-idempotency.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 45966,
          "output_tokens": 1212,
          "cached_tokens": 33895,
          "wall_ms": 21361
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.84,
          "sd": 0.020276,
          "judge_sd_mean": 0.013631,
          "variance_ratio": 1.487492,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "3a3f2dfdad91c12a01e7d039023555bb556709ada475745e5ac6599499cbefc2",
              "status": "measured",
              "samples": [
                0.78,
                0.85,
                0.82
              ],
              "judge_sample_hashes": [
                "a1d8ffb867680ee1894a7f7bee481fa26dcb718e2be76209bdd663751f92b3c8",
                "e6bd23e13f6869ba4ebb227b4edbe1ff57e131bf8170389b9e18239e126e0146",
                "6ea4086c90d3ce8982ae0065ec1f5c2590ed338a7a452c4a994defef00ed8c2a"
              ],
              "mean": 0.816667,
              "stddev": 0.035119,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 16253,
                "output_tokens": 953,
                "cached_tokens": 540,
                "wall_ms": 16745
              },
              "judge_usage": {
                "input_tokens": 45966,
                "output_tokens": 1854,
                "cached_tokens": 33895,
                "wall_ms": 30727
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "6138d2d937d270187d98ee2c1be9c393f7f24d5fce1d0624a6ca8445851bb777",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "1e02a824de7df3deb25f55bcc3bfbda93608ff9bbda5a6a9f394c9bc515e40c5",
                "be42ff3436e0a70c7cee1e6174d45e30eda9a3022bd43a9ad71d87831b2bbe2c",
                "5da41b36bd3b395356a0012fd03df0cec43dd446df5ed1267bbec2b3446e4e24"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 16253,
                "output_tokens": 1303,
                "cached_tokens": 16251,
                "wall_ms": 15808
              },
              "judge_usage": {
                "input_tokens": 46557,
                "output_tokens": 1443,
                "cached_tokens": 34289,
                "wall_ms": 24837
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "43bf2a71ada321bc7384bb161dcad24d62beb0cc5293008be6b31ad9dcacbebe",
              "status": "measured",
              "samples": [
                0.86,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "b68ec3b27be90469cfeb3c5146eece5a41ff28233d039a4dc56461a7c5c1dc67",
                "56b2aa39842058989a00501fb038a37d44577edc2cbfc7db8e7f058c88edc182",
                "c2b92cad5d4d2673541594d44e37792b7fda19f6ce265542c0b209065d83867b"
              ],
              "mean": 0.853333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 16253,
                "output_tokens": 1057,
                "cached_tokens": 16251,
                "wall_ms": 13383
              },
              "judge_usage": {
                "input_tokens": 45966,
                "output_tokens": 1212,
                "cached_tokens": 33895,
                "wall_ms": 21361
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.517778,
        "mean": 0.517778,
        "stddev": 0.083178,
        "samples": [
          0.55,
          0.6,
          0.6
        ],
        "generation_hash": "b1ae01fd0fe07126dd4d38336aef64298c06c329d04bc20f3b5daaa63e3d7244",
        "judge_sample_hashes": [
          "946ba53ab55f2ed48f749230bd484de88ea7216a94ae1f20069d2e3b7ee5ad86",
          "4eba966dff527a20d64dd2d8f1f9df1cfbe646e9ddc6aff824228a5a60537677",
          "93d8be64893b77ed9b73e78fcebfdafcf70dc2cf8539593bfc0b56a09d74a13f"
        ],
        "threshold": 0.7,
        "reason": "Met (a), (b), and largely (d) with useful mutation/intent notes, but left the bare TODO dangling and the 'Gold tier gets 10% off' line partly restates the code.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 45327,
          "output_tokens": 1922,
          "cached_tokens": 33469,
          "wall_ms": 31262
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 6,
          "n_measured": 6,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.517778,
          "sd": 0.083178,
          "judge_sd_mean": 0.037421,
          "variance_ratio": 2.222763,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "8f7b25980313bf4a194f6d6022b2a7ddacf2c4f049b72cb323c99f4d9df8827e",
              "status": "measured",
              "samples": [
                0.4,
                0.42,
                0.4
              ],
              "judge_sample_hashes": [
                "ebfd88a776f34bcd849797e1180cb5377a088db4d7135516df72a8776e673d4d",
                "46864fce11db8ad6127ac5acf2b61da72981232091ea57b75a4ceffcad191c8c",
                "17b07338f6ff769dc25037db3a76369862bb21a9a0f0ba97e388e15107583fce"
              ],
              "mean": 0.406667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12790,
                "output_tokens": 827,
                "cached_tokens": 2144,
                "wall_ms": 10813
              },
              "judge_usage": {
                "input_tokens": 45699,
                "output_tokens": 1888,
                "cached_tokens": 33717,
                "wall_ms": 31652
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "9a1c836e4720ea073e6b90c95fa49a0da0dcb3b9a416c636b3fac6d713a7747e",
              "status": "measured",
              "samples": [
                0.45,
                0.4,
                0.45
              ],
              "judge_sample_hashes": [
                "88d512c2e7dc8930bf5631bc7a53b4d5dc92649f5fb877750a674642e7dc657d",
                "0085832edb8802c52eadd8d39115d18095e1e07e684942862a77024603dc140c",
                "348f75f21047b215e3783d4237566a0bdb4b0e999420d9d52c7b0a823f43c3c7"
              ],
              "mean": 0.433333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12790,
                "output_tokens": 811,
                "cached_tokens": 12788,
                "wall_ms": 10961
              },
              "judge_usage": {
                "input_tokens": 45735,
                "output_tokens": 3053,
                "cached_tokens": 33741,
                "wall_ms": 44982
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "2342335ca8a3636a33f570bc44d828c853a0c65fb38d0abdcad000ceaed56daa",
              "status": "measured",
              "samples": [
                0.5,
                0.6,
                0.55
              ],
              "judge_sample_hashes": [
                "b7ef4d5ffabde72717fd3444ba7560b038e779b8ea0c5a3abbc40638bfb0335f",
                "97de6e384655397805edfc6e4f5aa35b4c9e24c12b59f5e670af37b35948e3db",
                "da41e2fe1532f5144b93a3e8ac95f2f7ec18395816ca4e1af344100fe491a8d0"
              ],
              "mean": 0.55,
              "stddev": 0.05,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12790,
                "output_tokens": 783,
                "cached_tokens": 12788,
                "wall_ms": 10178
              },
              "judge_usage": {
                "input_tokens": 45714,
                "output_tokens": 2602,
                "cached_tokens": 33727,
                "wall_ms": 44073
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "adb0919f0d12dc22625df5deae24bcf89e7af37a33774fb07db9c7059292d8e9",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.65
              ],
              "judge_sample_hashes": [
                "74825a65d6f6bfdd2775dd3b62e0a34e28c99b7c170678134281c8d10c294573",
                "6ed16441c8b382828ee173c97d569887b4b4dd8bcac0aaf9a4312bc0193cbaca",
                "2a6fb0f831fe377db2f4d5d5e159f39b1e6acf22fbbdced5474034c2c971e533"
              ],
              "mean": 0.616667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12790,
                "output_tokens": 856,
                "cached_tokens": 12788,
                "wall_ms": 11065
              },
              "judge_usage": {
                "input_tokens": 45765,
                "output_tokens": 2033,
                "cached_tokens": 33761,
                "wall_ms": 33419
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "079e7f47f5ab8d31ca1f6065435e6e503e98431544033d33782b3711443e4893",
              "status": "measured",
              "samples": [
                0.45,
                0.6,
                0.5
              ],
              "judge_sample_hashes": [
                "78d83b1818a2397f497c31b42c0e20ccf2881c8245ff7ddac2ceaea099b620a1",
                "590f692f46507fdeba565af6b6c67f7a54b89c8295e9c3bd46a7ce733116b2ab",
                "e49b4d36abd8000962741e19a6f37b7bd1333d5b80686eb200ff03b881b5163b"
              ],
              "mean": 0.516667,
              "stddev": 0.076376,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12790,
                "output_tokens": 696,
                "cached_tokens": 12788,
                "wall_ms": 10158
              },
              "judge_usage": {
                "input_tokens": 45483,
                "output_tokens": 2268,
                "cached_tokens": 33573,
                "wall_ms": 38328
              }
            },
            {
              "draw_index": 5,
              "generation_hash": "b1ae01fd0fe07126dd4d38336aef64298c06c329d04bc20f3b5daaa63e3d7244",
              "status": "measured",
              "samples": [
                0.55,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "946ba53ab55f2ed48f749230bd484de88ea7216a94ae1f20069d2e3b7ee5ad86",
                "4eba966dff527a20d64dd2d8f1f9df1cfbe646e9ddc6aff824228a5a60537677",
                "93d8be64893b77ed9b73e78fcebfdafcf70dc2cf8539593bfc0b56a09d74a13f"
              ],
              "mean": 0.583333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12790,
                "output_tokens": 642,
                "cached_tokens": 12788,
                "wall_ms": 8902
              },
              "judge_usage": {
                "input_tokens": 45327,
                "output_tokens": 1922,
                "cached_tokens": 33469,
                "wall_ms": 31262
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.84,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.517778,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.84,
    "baseline_score": 0.517778,
    "delta": 0.322222,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "be8805bb87c940013f61bbf44a6593ead27cb884728b468f6ad9fb79006f1279",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 16253,
      "mean_output_tokens": 1104.33,
      "mean_cost_usd_per_call": 0.087099,
      "median_wall_ms": 15808,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 6,
      "mean_input_tokens": 12790,
      "mean_output_tokens": 769.17,
      "mean_cost_usd_per_call": 0.066543,
      "median_wall_ms": 10495.5,
      "wall_ms_p25": 10158,
      "wall_ms_p75": 10961,
      "wall_ms_iqr": 803
    },
    "skill_incremental_cost_usd_per_call": 0.020556,
    "skill_incremental_cost_usd_per_1k_calls": 20.556,
    "output_tokens_delta": 335.16,
    "median_wall_ms_delta": 5312.5,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.534815,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}