{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-opus-5",
    "model_release_date": "2026-07-24",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-15T16:11:24.010Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5",
      "reported_models": [
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-15T16:11:24.010Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.822222,
        "mean": 0.822222,
        "stddev": 0.016777,
        "samples": [
          0.83,
          0.85,
          0.84
        ],
        "generation_hash": "3fadf4e864b4837722be9ce897e5f12b8f4a8164bd145faceef38a0e2ceb995f",
        "judge_sample_hashes": [
          "68e895453a128da90884a2b398edf9e33213a1d11e4a9ba5e053a2f98ddf1a2e",
          "937e847fa8ebee3e726dbe98252c147090a1a4925b3a9cc80c19fe11e65ba4b0",
          "0da48c53f786bedce460cdc040b715ea536fe6193645fd64c8cb707f27ec0249"
        ],
        "threshold": 0.7,
        "reason": "Meets (a), (b), and (c). Added comments give non-obvious caller intent (side effects, total not recomputed), though the JSDoc partly narrates behavior and omits why gold gets 10%.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 12468,
          "output_tokens": 993,
          "cached_tokens": 10467,
          "wall_ms": 21672
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.822222,
          "sd": 0.016777,
          "judge_sd_mean": 0.013849,
          "variance_ratio": 1.211423,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "2a3c5564d7eb504f0d414fe3c4df271d6f92245958259b51d2643670e9ef8f40",
              "status": "measured",
              "samples": [
                0.8,
                0.82,
                0.84
              ],
              "judge_sample_hashes": [
                "760689c91a75ccb11a33fc2fe8e6275e258cb3909cf0792c65bede7444b84a2a",
                "11ce33b4874db85447b432bb2368565d6dd926ba0c2f1c4badfc2743d1ace49d",
                "e50a55949803e74cdb2a0f0acdb10c6fc7bc5b632925afeaa35bbc2cd0f4fea9"
              ],
              "mean": 0.82,
              "stddev": 0.02,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 6077,
                "output_tokens": 1095,
                "cached_tokens": 0,
                "wall_ms": 18266
              },
              "judge_usage": {
                "input_tokens": 12753,
                "output_tokens": 1023,
                "cached_tokens": 10657,
                "wall_ms": 22232
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "e6fca9b27b136acc9893f12f2d4d55e40fffe3f109fb00b3e60b47b0e0a5feaf",
              "status": "measured",
              "samples": [
                0.8,
                0.82,
                0.8
              ],
              "judge_sample_hashes": [
                "f52f94eb720ebfb4cb146592851805ad253ddba699c947e0375f83d27f7195ea",
                "865abcbcdffe61d704d28aa73cfd0c2ffc89335745beeaea58b0412c30a70f03",
                "1841e048c657cdec2ea5dbd623291a8d4ffde37a48405bd8edade511bf08cfa5"
              ],
              "mean": 0.806667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 6077,
                "output_tokens": 1279,
                "cached_tokens": 6075,
                "wall_ms": 20207
              },
              "judge_usage": {
                "input_tokens": 12444,
                "output_tokens": 995,
                "cached_tokens": 10451,
                "wall_ms": 20908
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "3fadf4e864b4837722be9ce897e5f12b8f4a8164bd145faceef38a0e2ceb995f",
              "status": "measured",
              "samples": [
                0.83,
                0.85,
                0.84
              ],
              "judge_sample_hashes": [
                "68e895453a128da90884a2b398edf9e33213a1d11e4a9ba5e053a2f98ddf1a2e",
                "937e847fa8ebee3e726dbe98252c147090a1a4925b3a9cc80c19fe11e65ba4b0",
                "0da48c53f786bedce460cdc040b715ea536fe6193645fd64c8cb707f27ec0249"
              ],
              "mean": 0.84,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 6077,
                "output_tokens": 1122,
                "cached_tokens": 6075,
                "wall_ms": 18759
              },
              "judge_usage": {
                "input_tokens": 12468,
                "output_tokens": 993,
                "cached_tokens": 10467,
                "wall_ms": 21672
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.531667,
        "mean": 0.531667,
        "stddev": 0.092756,
        "samples": [
          0.6,
          0.6,
          0.6
        ],
        "generation_hash": "ff3f8a3097f8004daca40b685ba0307f314b72a7308f62cf8b0238e4cd8c7889",
        "judge_sample_hashes": [
          "d96b28100e6644f17789821d01336db80c6257866342f672fa38c037dfc28fdf",
          "0588ad4b6eaa28024532fa1564c259f4f41b76a2cbd0ec5dd27f0e86fc7b8254",
          "a1657fdafac3a3cb4fd9f13942a2eba705e2062153ada19074f249fd3db9901f"
        ],
        "threshold": 0.7,
        "reason": "Meets (a) and (b), and (d) holds with intent-focused added comments; misses (c): the expired-coupons TODO is reworded but still an unhandled TODO, not implemented or removed.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 12399,
          "output_tokens": 1753,
          "cached_tokens": 10421,
          "wall_ms": 30822
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.531667,
          "sd": 0.092756,
          "judge_sd_mean": 0.010104,
          "variance_ratio": 9.180127,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "b1e728f7b6a6f9529454073e3e0b8a85c31a0dc3b63d09bf5fe806ac6845aa2c",
              "status": "measured",
              "samples": [
                0.6,
                0.58,
                0.6
              ],
              "judge_sample_hashes": [
                "bc65780d6fd0767813d1bd7758cec78eb0eaecede0e6c884ca4d4849fc38baa1",
                "78ce2b3d7c75403034edd3f35281c17c1489e8d2599b70a9b841cd50c85a1eaa",
                "8660f883e8599bee13810eb9ee12ebb7cba4e8ea15aa45661f9fc90964abf163"
              ],
              "mean": 0.593333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 2614,
                "output_tokens": 889,
                "cached_tokens": 2053,
                "wall_ms": 13933
              },
              "judge_usage": {
                "input_tokens": 12312,
                "output_tokens": 1379,
                "cached_tokens": 10363,
                "wall_ms": 25618
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "0875e515d54ec40f0e90ae48a5961faf90328e9cc819a02f245beb68323397a1",
              "status": "measured",
              "samples": [
                0.55,
                0.55,
                0.5
              ],
              "judge_sample_hashes": [
                "75c67b3eb173e11150d94833183592d404c1bbe2693075ee32762235be6a3b53",
                "dfda813571ec1692432ce7401791c9ee8fd9c52f59cc9aa44abc4c53691ae8e6",
                "0641680414961494ca9ce50b7d615138f283d2860a48e3ac042452369100f143"
              ],
              "mean": 0.533333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 2614,
                "output_tokens": 974,
                "cached_tokens": 2612,
                "wall_ms": 15360
              },
              "judge_usage": {
                "input_tokens": 12375,
                "output_tokens": 1193,
                "cached_tokens": 10405,
                "wall_ms": 23513
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "e248ec7baf67e1ed736095abc0b9e99eeaf2bfdaf7ff07dce47c2914a964f10f",
              "status": "measured",
              "samples": [
                0.4,
                0.4,
                0.4
              ],
              "judge_sample_hashes": [
                "fea0cee32fb638e24daa5977185aa299c25abb6b2c22f074ba6cac439a11728e",
                "e7400b0f2522123d3ee5f155605a917fb0feafb351a806fa6990e7338143188f",
                "f2de4548bf6a392769b57c486108cb84f0ae590c5f3c847ab516c847fb344aa0"
              ],
              "mean": 0.4,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 2614,
                "output_tokens": 881,
                "cached_tokens": 2612,
                "wall_ms": 14539
              },
              "judge_usage": {
                "input_tokens": 12255,
                "output_tokens": 667,
                "cached_tokens": 10325,
                "wall_ms": 17461
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "ff3f8a3097f8004daca40b685ba0307f314b72a7308f62cf8b0238e4cd8c7889",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "d96b28100e6644f17789821d01336db80c6257866342f672fa38c037dfc28fdf",
                "0588ad4b6eaa28024532fa1564c259f4f41b76a2cbd0ec5dd27f0e86fc7b8254",
                "a1657fdafac3a3cb4fd9f13942a2eba705e2062153ada19074f249fd3db9901f"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 2614,
                "output_tokens": 945,
                "cached_tokens": 2612,
                "wall_ms": 15055
              },
              "judge_usage": {
                "input_tokens": 12399,
                "output_tokens": 1753,
                "cached_tokens": 10421,
                "wall_ms": 30822
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.822222,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.531667,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.822222,
    "baseline_score": 0.531667,
    "delta": 0.290555,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "3f52ddb6b849ee54d1ea4e137f7f797a3f6b58f8f38f05ee8442a6a48321f50b",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 6077,
      "mean_output_tokens": 1165.33,
      "mean_cost_usd_per_call": 0.059518,
      "median_wall_ms": 18759,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 4,
      "mean_input_tokens": 2614,
      "mean_output_tokens": 922.25,
      "mean_cost_usd_per_call": 0.036126,
      "median_wall_ms": 14797,
      "wall_ms_p25": 14236,
      "wall_ms_p75": 15207.5,
      "wall_ms_iqr": 971.5
    },
    "skill_incremental_cost_usd_per_call": 0.023392,
    "skill_incremental_cost_usd_per_1k_calls": 23.392,
    "output_tokens_delta": 243.08,
    "median_wall_ms_delta": 3962,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.192985,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}