{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-sonnet-5",
    "model_release_date": "2026-06-30",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T05:04:47.416Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T05:04:47.416Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "borderline",
        "score": 0.7,
        "mean": 0.7,
        "stddev": 0.098432,
        "samples": [
          0.6,
          0.6,
          0.6
        ],
        "generation_hash": "1a6398c07ed4500d4729e1666c64301b6fd4bc1859455129744b6bfe1ff2f46f",
        "judge_sample_hashes": [
          "dc16578544fe29804eb614566b9898cf343a9ab51058e7197271283f1713a9a0",
          "641f0ca286aac5d5b1b6df9b6b5ece89f4074f5c964ae9f731d46759c641c6c4",
          "624343067b49ff13cc4e78a2bfa5ac5861514d1d78f196bb5ed81b393bb69555"
        ],
        "threshold": 0.7,
        "reason": "Met (a) and (b) and added no 'what' comments, but left the bare TODO dangling (c) and added no intent/why comment for the gold-tier discount (d).",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44298,
          "output_tokens": 1170,
          "cached_tokens": 32783,
          "wall_ms": 23668
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 5,
          "n_measured": 5,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.7,
          "sd": 0.098432,
          "judge_sd_mean": 0.004619,
          "variance_ratio": 21.31024,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "7c5cb46a2471e9397e483c7eadafa5eac449ebf00d2132f99209d30a0ef50ef1",
              "status": "measured",
              "samples": [
                0.8,
                0.8,
                0.8
              ],
              "judge_sample_hashes": [
                "89037d40bb9747a6950c76d7b7bae55c29f39b2e338d561b5c262cdb8995af94",
                "d617f11300be44632ba43f9f880d30d5391ddc526e853da7ada9850a9277d970",
                "be847a7884d959df539ab69ac5fdbc92ad3f1bbcc69398390d6506059fd49446"
              ],
              "mean": 0.8,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 297,
                "cached_tokens": 22673,
                "wall_ms": 5101
              },
              "judge_usage": {
                "input_tokens": 44319,
                "output_tokens": 454,
                "cached_tokens": 32797,
                "wall_ms": 24882
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "979c2c49c592ef78006c9ec397ab75093025c73d8165405f56e2ae4995edcdc3",
              "status": "measured",
              "samples": [
                0.6,
                0.62,
                0.6
              ],
              "judge_sample_hashes": [
                "d62c8694b969d8fa9e4429d2eefb8dd093b32657c002e00b69d534165877f687",
                "9d8e210c6ec155a6cb63db3efc9c8fed3fe5fb094cbfcda0c2c6f8b34edbb822",
                "064c95a7c9dd068348b644ee5773ec5e2242e24d51998f8b96f67ec35cc631bd"
              ],
              "mean": 0.606667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 374,
                "cached_tokens": 22673,
                "wall_ms": 6388
              },
              "judge_usage": {
                "input_tokens": 44886,
                "output_tokens": 1448,
                "cached_tokens": 33175,
                "wall_ms": 37638
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "2837b957325f27a5624967b7b74a5ebfadcdf374d347c0b373c721b5ec561534",
              "status": "measured",
              "samples": [
                0.8,
                0.8,
                0.8
              ],
              "judge_sample_hashes": [
                "6a3299da1c1f7076233b2e6eb37d282d5ccb3cfc0816a5904fcc01d10d70a891",
                "1e33c8c3b5d5c2572902572fbd6a4c8f5fe2f24d1bf938a76237545756bd4aa2",
                "3fdc3e2d4e6fc12cd546aeed5520bab9020ee21096a96990d1b2d8dcf3d27beb"
              ],
              "mean": 0.8,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 315,
                "cached_tokens": 22673,
                "wall_ms": 6840
              },
              "judge_usage": {
                "input_tokens": 44379,
                "output_tokens": 840,
                "cached_tokens": 32837,
                "wall_ms": 23937
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "ec223576ac3d8aae4a72ff9801f57f85dbf5aaa1d8fb184e02585da2a3d10a60",
              "status": "measured",
              "samples": [
                0.68,
                0.7,
                0.7
              ],
              "judge_sample_hashes": [
                "834b6af8d1595d93d5090087a9c5d6a47cdfa22732534741c86c5c6acaeed3a1",
                "173044787854c050cae8522473d8c4223b5bf18db5ab95c96527b6eaa85bd93d",
                "a88cf5ca3f2a1e0d383d627d2301de91b9e269f4f4acef2d1a5e341694090cca"
              ],
              "mean": 0.693333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 661,
                "cached_tokens": 22673,
                "wall_ms": 8701
              },
              "judge_usage": {
                "input_tokens": 44463,
                "output_tokens": 1199,
                "cached_tokens": 32893,
                "wall_ms": 27291
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "1a6398c07ed4500d4729e1666c64301b6fd4bc1859455129744b6bfe1ff2f46f",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "dc16578544fe29804eb614566b9898cf343a9ab51058e7197271283f1713a9a0",
                "641f0ca286aac5d5b1b6df9b6b5ece89f4074f5c964ae9f731d46759c641c6c4",
                "624343067b49ff13cc4e78a2bfa5ac5861514d1d78f196bb5ed81b393bb69555"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 1077,
                "cached_tokens": 22673,
                "wall_ms": 14921
              },
              "judge_usage": {
                "input_tokens": 44298,
                "output_tokens": 1170,
                "cached_tokens": 32783,
                "wall_ms": 23668
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "borderline",
        "score": 0.706,
        "mean": 0.706,
        "stddev": 0.104254,
        "samples": [
          0.6,
          0.58,
          0.6
        ],
        "generation_hash": "965f234b88eca35ab522f03e7596512c11cd46add65cbb72b2fea5947a8b6ec5",
        "judge_sample_hashes": [
          "3f0c549d4c065e8521c0d7d2df52b29954819b1806c3c75f8d49aa02ed4ca306",
          "542940364011c31b90d950241a4c9e7741cfd727dcc1ddf31bd25ab7797434b1",
          "48cccc96dad48ec1a3bb2898c1e505ebf9fab49e061611b70230f514da6333d9"
        ],
        "threshold": 0.7,
        "reason": "Meets (a) and (b) — all three restating comments and the commented-out legacyRate line removed — but leaves the bare TODO dangling, missing (c); no intent comment added per (d).",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44424,
          "output_tokens": 1309,
          "cached_tokens": 32867,
          "wall_ms": 27819
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 5,
          "n_measured": 5,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.706,
          "sd": 0.104254,
          "judge_sd_mean": 0.019555,
          "variance_ratio": 5.331322,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "55ee1cb672e182458dd3dbbe2c13326f4942e0f1d2296906d193ab268ee90031",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "966d620c4f6e9a5a9a2ee12f03873f0b945e565af6a2cb7feb97d9b85cf8e0e6",
                "e476795d4685b25c6643e45aff2143bfa1db1bba627a74df3dfb19b6c73617db",
                "b6dc9334957ece48ecb12d08339f890539a776d36fc5c979fbc5dd34ff7b6cd5"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 485,
                "cached_tokens": 19210,
                "wall_ms": 9437
              },
              "judge_usage": {
                "input_tokens": 44328,
                "output_tokens": 1587,
                "cached_tokens": 32803,
                "wall_ms": 28114
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "db0be932df65d003146be999b7faad30a30e422a0c0e82b5c3791222916d87f7",
              "status": "measured",
              "samples": [
                0.8,
                0.8,
                0.8
              ],
              "judge_sample_hashes": [
                "00e930a5777debd3957392d42d14bf34b494f06551a646311ab1b8e2ca2118e4",
                "59d1040d9d70a017364493490f2b61c2540a1c8bce4954f3bcfc057ab532c660",
                "96ae67d6007b8631c41459fdb1c39119438b3671ed20388aa7254227ddbfb117"
              ],
              "mean": 0.8,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 507,
                "cached_tokens": 19210,
                "wall_ms": 8201
              },
              "judge_usage": {
                "input_tokens": 44373,
                "output_tokens": 546,
                "cached_tokens": 32833,
                "wall_ms": 20032
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "e3daf2ae305818c70165fc8ac15af9e6392808c9f2babcf629f148f8034dfd44",
              "status": "measured",
              "samples": [
                0.84,
                0.8,
                0.78
              ],
              "judge_sample_hashes": [
                "6dc398a33cbe385aee3871c974071e441ecc18bb3a1679a0f183b04e1fa93989",
                "d65b58b6c15f77c73a8dfca45f6ec7e28ba8d268cc188cb19072ebabb696a19b",
                "ad8c210b7989c7fe56913599924c0d2d2f3aa82197a7c93b7d787ad3304fc370"
              ],
              "mean": 0.806667,
              "stddev": 0.030551,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 401,
                "cached_tokens": 19210,
                "wall_ms": 8404
              },
              "judge_usage": {
                "input_tokens": 44967,
                "output_tokens": 1828,
                "cached_tokens": 33229,
                "wall_ms": 41973
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "a1a2798b3edef0a5eaaa0490b59c17248d3860dc1b3f4eb85cb243c0b7244243",
              "status": "measured",
              "samples": [
                0.72,
                0.68,
                0.79
              ],
              "judge_sample_hashes": [
                "4054290522c9d539542246e27f4bf1be3e24f0c9732e6aa2a7a98256c25c2277",
                "c52f11c5f347c8e90f1b8bd309c89ae8a3bf564dde94ae1bb2c2fd955169da95",
                "1b2a2839991a40618a472930e39d07071d35c7901ee71143b1f3f778e39d050c"
              ],
              "mean": 0.73,
              "stddev": 0.055678,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 248,
                "cached_tokens": 19210,
                "wall_ms": 5323
              },
              "judge_usage": {
                "input_tokens": 44508,
                "output_tokens": 1395,
                "cached_tokens": 32923,
                "wall_ms": 27157
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "965f234b88eca35ab522f03e7596512c11cd46add65cbb72b2fea5947a8b6ec5",
              "status": "measured",
              "samples": [
                0.6,
                0.58,
                0.6
              ],
              "judge_sample_hashes": [
                "3f0c549d4c065e8521c0d7d2df52b29954819b1806c3c75f8d49aa02ed4ca306",
                "542940364011c31b90d950241a4c9e7741cfd727dcc1ddf31bd25ab7797434b1",
                "48cccc96dad48ec1a3bb2898c1e505ebf9fab49e061611b70230f514da6333d9"
              ],
              "mean": 0.593333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 220,
                "cached_tokens": 19210,
                "wall_ms": 5221
              },
              "judge_usage": {
                "input_tokens": 44424,
                "output_tokens": 1309,
                "cached_tokens": 32867,
                "wall_ms": 27819
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.7,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.706,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.7,
    "baseline_score": 0.706,
    "delta": -0.006,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "be6e3ee15add531093f701dd6249e7bbb4392ec3a8a861092eb8fb5700539c50",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 5,
      "mean_input_tokens": 22675,
      "mean_output_tokens": 544.8,
      "mean_cost_usd_per_call": 0.050798,
      "median_wall_ms": 6840,
      "wall_ms_p25": 5744.5,
      "wall_ms_p75": 11811,
      "wall_ms_iqr": 6066.5
    },
    "baseline": {
      "call_count": 5,
      "mean_input_tokens": 19212,
      "mean_output_tokens": 372.2,
      "mean_cost_usd_per_call": 0.042146,
      "median_wall_ms": 8201,
      "wall_ms_p25": 5272,
      "wall_ms_p75": 8920.5,
      "wall_ms_iqr": 3648.5
    },
    "skill_incremental_cost_usd_per_call": 0.008652,
    "skill_incremental_cost_usd_per_1k_calls": 8.652,
    "output_tokens_delta": 172.6,
    "median_wall_ms_delta": -1361,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.505585,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}