{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-opus-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-23T06:43:12.011Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-opus-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-23T06:43:12.011Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5-5": {
          "input_per_mtok": 4,
          "output_per_mtok": 20,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "borderline",
        "score": 0.661111,
        "mean": 0.661111,
        "stddev": 0.147764,
        "samples": [
          0.45,
          0.6,
          0.55
        ],
        "generation_hash": "f4ed0167d112d4c095315111c9764efddc5e05344aa2e86d52dd4b8cc1f723e2",
        "judge_sample_hashes": [
          "032da96e3bce77b2840bbb6774efd3f0fd34c475c015aa808f049e3e1b914216",
          "65643084b3e59ea4997e57b0d241e4d792725f32e6eeb027abd3934bd97b3a6b",
          "b8aefdde7794ba5312ec1bfc7484ab0345fdc680c985516123a5cbf57ab1a98e"
        ],
        "threshold": 0.7,
        "reason": "Met (a) and (b) cleanly, but left the TODO as a restatement with an admitted placeholder reason, and the added '10% gold-tier discount' comment restates the what, not why.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 18585,
          "output_tokens": 2351,
          "cached_tokens": 15642,
          "wall_ms": 35746
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 6,
          "n_measured": 6,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.661111,
          "sd": 0.147764,
          "judge_sd_mean": 0.043744,
          "variance_ratio": 3.377926,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "c7276cb74f981e1cc536279f25197b31c4aefb2fdc0dd917e187f8ebff785457",
              "status": "measured",
              "samples": [
                0.5,
                0.5,
                0.45
              ],
              "judge_sample_hashes": [
                "c4840ababc1337b8aa949d8bfcb8be79573e3c5876c94745ba15b3f40f4924af",
                "20799eaa70687fd0c7745a309b4df99e25c14af6c20f07680f74c874247fc9a5",
                "e6fa0ca66c12f1a09f67b04225d8bacb1ddb8519ec0d51f9f5f6e9628a4cd941"
              ],
              "mean": 0.483333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 7187,
                "output_tokens": 840,
                "cached_tokens": 540,
                "wall_ms": 10773
              },
              "judge_usage": {
                "input_tokens": 18483,
                "output_tokens": 2692,
                "cached_tokens": 15574,
                "wall_ms": 39014
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "283e07542179f8f516e1486ff00010a649c9a06f8024bd1c26b819605dab2901",
              "status": "measured",
              "samples": [
                0.82,
                0.75,
                0.83
              ],
              "judge_sample_hashes": [
                "79264c1d626d84039d4f838e84d272ebeccaf6fafaf89fa1ad9fdf9e6d2cad4a",
                "bc307e9f8ba9f66000594090e81a5e6e1ad4492b4ba4cbc195dd951f1321ef9a",
                "41fd2422ff2f5932c46bb2948f98d884cc56d7bf3e81e3fff52afee48e90c065"
              ],
              "mean": 0.8,
              "stddev": 0.043589,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 7187,
                "output_tokens": 808,
                "cached_tokens": 7185,
                "wall_ms": 10607
              },
              "judge_usage": {
                "input_tokens": 18453,
                "output_tokens": 1970,
                "cached_tokens": 15554,
                "wall_ms": 30210
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "b2d75929af9576560652cf901b7a7234f8896dca79bd19d041fda0dbcada8bb4",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "01be07d049ac6a3413eccdf07143b94cedd67198f37c692df659b0261df1f302",
                "620903380469cccad640b73d91ce689acf6b9cf48a6e1d6bce5ab8aed3176d55",
                "94b61af9cb94a047173e0a4e4dfad159f76b1b8c55c2a996644a5886c846fcdc"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 7187,
                "output_tokens": 829,
                "cached_tokens": 7185,
                "wall_ms": 10878
              },
              "judge_usage": {
                "input_tokens": 18237,
                "output_tokens": 1595,
                "cached_tokens": 15410,
                "wall_ms": 25547
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "7a9065550fdd632dbd133d985a10df4888dc7736b1a8b01783853d4ddf87b9b1",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.86
              ],
              "judge_sample_hashes": [
                "450a29d7e0568670b44219faea498bf57d0ffc7d28743b1c09a9258f207ca70d",
                "380645eec31218ee6f0837a139acc13fcb37c650bf429436ffbe0f39ad6cf663",
                "bffeb3ddc40dbbc500c37c9a698e4f0dec26c57ed548e2233989aefbb0544882"
              ],
              "mean": 0.853333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 7187,
                "output_tokens": 936,
                "cached_tokens": 7185,
                "wall_ms": 12048
              },
              "judge_usage": {
                "input_tokens": 18693,
                "output_tokens": 1288,
                "cached_tokens": 15714,
                "wall_ms": 21504
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "4b5b0738272a3852070841c4da086f8f3e71a08cf8daf7629aa8a2e45101bad6",
              "status": "measured",
              "samples": [
                0.62,
                0.82,
                0.65
              ],
              "judge_sample_hashes": [
                "62731c9f9db32d7f4c66c463316f353c8125e7e3c54e71c1545fec9875112914",
                "e0a131283ac238db5d5dcdc37ce28236231b5a765cf616730fe5e544352e9d9b",
                "9a4658fcf45c00ea21d5cd023f9138ab9f379e89e26132e8b1ab99f8a3c91966"
              ],
              "mean": 0.696667,
              "stddev": 0.107858,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 7187,
                "output_tokens": 889,
                "cached_tokens": 7185,
                "wall_ms": 10835
              },
              "judge_usage": {
                "input_tokens": 18732,
                "output_tokens": 2332,
                "cached_tokens": 15740,
                "wall_ms": 34150
              }
            },
            {
              "draw_index": 5,
              "generation_hash": "f4ed0167d112d4c095315111c9764efddc5e05344aa2e86d52dd4b8cc1f723e2",
              "status": "measured",
              "samples": [
                0.45,
                0.6,
                0.55
              ],
              "judge_sample_hashes": [
                "032da96e3bce77b2840bbb6774efd3f0fd34c475c015aa808f049e3e1b914216",
                "65643084b3e59ea4997e57b0d241e4d792725f32e6eeb027abd3934bd97b3a6b",
                "b8aefdde7794ba5312ec1bfc7484ab0345fdc680c985516123a5cbf57ab1a98e"
              ],
              "mean": 0.533333,
              "stddev": 0.076376,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 7187,
                "output_tokens": 841,
                "cached_tokens": 7185,
                "wall_ms": 10882
              },
              "judge_usage": {
                "input_tokens": 18585,
                "output_tokens": 2351,
                "cached_tokens": 15642,
                "wall_ms": 35746
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.566667,
        "mean": 0.566667,
        "stddev": 0,
        "samples": [
          0.55,
          0.55,
          0.6
        ],
        "generation_hash": "294a63179dd624e32b3121cb9b74d9eb7d738ee60ddbccc783713c2cb8dad2db",
        "judge_sample_hashes": [
          "bfa10cb400f179edad384f9aa53dce17700dfbe0590dddeb6f79d1e15ed0575b",
          "5658b2d27f5e44334417598a5ddabd29b2ab8b70151cb2c09219082cc27b091f",
          "e45cac6dc6249336d70f7b20cb53df2e1ed0a9ce8b59899785bdc761b0e88529"
        ],
        "threshold": 0.7,
        "reason": "Meets (a) and (b) fully; misses (c) by rephrasing rather than handling or removing the TODO; gold-tier comment partly restates the 'what' before adding intent.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 18033,
          "output_tokens": 2088,
          "cached_tokens": 15274,
          "wall_ms": 31022
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.566667,
          "sd": 0,
          "judge_sd_mean": 0.028868,
          "variance_ratio": 0,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "c4fa9e2bf62d5df93b2ba7f86688d91002704b7d97726dfb3da799c082e36046",
              "status": "measured",
              "samples": [
                0.55,
                0.6,
                0.55
              ],
              "judge_sample_hashes": [
                "ae504a3a4163b673995aabcca0b5e54f6a4f93e3c2be3a0ab608e655c4930741",
                "8c44a59885a3abe71a85fc0553d830e8b9356a4782ea754c4e8de266f668f972",
                "c76af0ea048da054dc61e625b3dce2fa0e5e5ab3a69dbe9deb5f884d4b36d698"
              ],
              "mean": 0.566667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 3724,
                "output_tokens": 528,
                "cached_tokens": 2145,
                "wall_ms": 7973
              },
              "judge_usage": {
                "input_tokens": 17934,
                "output_tokens": 1841,
                "cached_tokens": 15208,
                "wall_ms": 28025
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "c2248b92091f2d510922c39791ea89516068f04782b71c69531a54ddfda88f90",
              "status": "measured",
              "samples": [
                0.55,
                0.55,
                0.6
              ],
              "judge_sample_hashes": [
                "2f3740e3ae2745df4956761f197aa2ea605cbcc514414f6c60d5dc45c765ec03",
                "c451927531b4ee125b0ec658899eb749ca6ee69c8577601fdda7d0792051f7bd",
                "e234d99047256b06984661c06f876c072f1de8643015740f03490a4c2a5567bb"
              ],
              "mean": 0.566667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 3724,
                "output_tokens": 721,
                "cached_tokens": 3722,
                "wall_ms": 9540
              },
              "judge_usage": {
                "input_tokens": 18201,
                "output_tokens": 2381,
                "cached_tokens": 15386,
                "wall_ms": 35180
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "294a63179dd624e32b3121cb9b74d9eb7d738ee60ddbccc783713c2cb8dad2db",
              "status": "measured",
              "samples": [
                0.55,
                0.55,
                0.6
              ],
              "judge_sample_hashes": [
                "bfa10cb400f179edad384f9aa53dce17700dfbe0590dddeb6f79d1e15ed0575b",
                "5658b2d27f5e44334417598a5ddabd29b2ab8b70151cb2c09219082cc27b091f",
                "e45cac6dc6249336d70f7b20cb53df2e1ed0a9ce8b59899785bdc761b0e88529"
              ],
              "mean": 0.566667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 3724,
                "output_tokens": 587,
                "cached_tokens": 3722,
                "wall_ms": 8364
              },
              "judge_usage": {
                "input_tokens": 18033,
                "output_tokens": 2088,
                "cached_tokens": 15274,
                "wall_ms": 31022
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.661111,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.566667,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.661111,
    "baseline_score": 0.566667,
    "delta": 0.094444,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "b924a435c70e0092453bc745941954ad08e12c8932e0d0120517151b44d29a15",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 6,
      "mean_input_tokens": 7187,
      "mean_output_tokens": 857.17,
      "mean_cost_usd_per_call": 0.045891,
      "median_wall_ms": 10856.5,
      "wall_ms_p25": 10773,
      "wall_ms_p75": 10882,
      "wall_ms_iqr": 109
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 3724,
      "mean_output_tokens": 612,
      "mean_cost_usd_per_call": 0.027136,
      "median_wall_ms": 8364,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.018755,
    "skill_incremental_cost_usd_per_1k_calls": 18.755,
    "output_tokens_delta": 245.17,
    "median_wall_ms_delta": 2492.5,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.294065,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}