{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-sonnet-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T04:51:21.171Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T04:51:21.171Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.813333,
        "mean": 0.813333,
        "stddev": 0.094458,
        "samples": [
          0.85,
          0.83,
          0.87
        ],
        "generation_hash": "cb2a1d7a53ac055864c0decbf9cdef651f4309f47a3f850e68c0bb1427d14d9b",
        "judge_sample_hashes": [
          "7d14082e506719becbaff3e47b98bcb7ac4e0c3cf151920dd0afe92dccee39fb",
          "b0ba017d435e2e9b2d86f60c22bcfc869bb629b1b068cb00e900932b9efaa910",
          "d2b13955db202c18e80038eb655b751f0ac5393217547fb9562edef9a1ec3047"
        ],
        "threshold": 0.7,
        "reason": "Removes all three restating comments (a), the commented-out legacyRate line (b), and the bare TODO (c); the single added comment names the gold-tier business rule behind 0.9 (d).",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 45351,
          "output_tokens": 1132,
          "cached_tokens": 33485,
          "wall_ms": 24655
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 7,
          "n_measured": 7,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.813333,
          "sd": 0.094458,
          "judge_sd_mean": 0.013066,
          "variance_ratio": 7.229297,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "12ad3a86e6918831f24c40c3ed338033c13135fc94c64c6fbc84d100fc85241c",
              "status": "measured",
              "samples": [
                0.86,
                0.87,
                0.85
              ],
              "judge_sample_hashes": [
                "8cf298ab47e7f2b1aebc846632aae0abe63fcd5acf3ab1bdfe8f160910c24936",
                "936ab43b50ff87293a9da18ede483321e5243f8e1ef17f9d9606cbd39177efcd",
                "ce052e79860553c116ddf95a463f89fcb99ef0c2ce26493661b319bfe9fd2aed"
              ],
              "mean": 0.86,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 469,
                "cached_tokens": 16255,
                "wall_ms": 6951
              },
              "judge_usage": {
                "input_tokens": 45171,
                "output_tokens": 409,
                "cached_tokens": 33365,
                "wall_ms": 14445
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "9000d3b8caf4637bdc17eb7316621aec6de0d4d7688d9c7beaf5efdf09f35eae",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "640469e5b0bd90ae7c9846135703c3d20cda251cbb2473a6b51f42077282eeb3",
                "d8113f5af0ed5d54b0f6a34945b0b858a0a42146ad4a20346101506799cc42c4",
                "f66b143ea25fdcc62a78a5c237588c2a06fdc80bdefce969d52a19c9ebd5ea4d"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 708,
                "cached_tokens": 16255,
                "wall_ms": 7335
              },
              "judge_usage": {
                "input_tokens": 45321,
                "output_tokens": 1494,
                "cached_tokens": 33465,
                "wall_ms": 36123
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "5e64cdf6620831e082497352f545f21ff743adc816553d5177cbb72393b400d1",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.84
              ],
              "judge_sample_hashes": [
                "ea3902e79e8e635236c9d3aacdf855908e934445d2b5ed4c01ce67b8bb9a3e77",
                "b11bc397a9efbff3d9e460ce1c5a370edff27362e6a513401d6598644861f2e7",
                "4f03c5f83c67f9dbbcaa01a4b03aed0139d61565b888c91365f2c16cb0f0cffc"
              ],
              "mean": 0.846667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 481,
                "cached_tokens": 16255,
                "wall_ms": 6042
              },
              "judge_usage": {
                "input_tokens": 45207,
                "output_tokens": 1436,
                "cached_tokens": 33389,
                "wall_ms": 27303
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "6367774c6fba72d90720964397d0937836bd7f4d8278fb08731c8179bfa91597",
              "status": "measured",
              "samples": [
                0.86,
                0.83,
                0.85
              ],
              "judge_sample_hashes": [
                "2e6022017fb8d67b6e4e5cf546111997eacab59e0f25eb722979951d888337ec",
                "c36926c9a645940839bafed8c0d99c5fbfffe64db3fcef2ced4be9d663de9881",
                "2c94251818b77eca4c9a074cd0a197e1678b5d7924160b7bf558ccf265257e50"
              ],
              "mean": 0.846667,
              "stddev": 0.015275,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 506,
                "cached_tokens": 16255,
                "wall_ms": 10293
              },
              "judge_usage": {
                "input_tokens": 45282,
                "output_tokens": 1138,
                "cached_tokens": 33439,
                "wall_ms": 21740
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "97ddfc6db15fa5e3908db41efc620cfb7435d3e1184362f0deca9fb347a5bb34",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.87
              ],
              "judge_sample_hashes": [
                "5e738e57da4998630be3a854a60f38b531f75d717b9c356625799c84e829e2b7",
                "d84e58a50bd77e0a2ebd8b1d5d9fdacc4eb1d873b64244dacf1f13cd7e028990",
                "2688fb01e7934d2f62dc1cb8fde2abeda3bce151d6cad753866585d22ad48b1c"
              ],
              "mean": 0.856667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 537,
                "cached_tokens": 16255,
                "wall_ms": 6108
              },
              "judge_usage": {
                "input_tokens": 45375,
                "output_tokens": 731,
                "cached_tokens": 33501,
                "wall_ms": 19562
              }
            },
            {
              "draw_index": 5,
              "generation_hash": "312afe141d4eaa4bcf6ba4149a8f8e7ef5b1e61881c74b6d2f267d1e04fc4f3c",
              "status": "measured",
              "samples": [
                0.8,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "5d92c27115b199249af0ddb0efcfe2477fef99ddf122e5c18fd019e795a13d1e",
                "29243657df224096df9a86ecbf00316ae52e1acaf9e2453f9bc466a060e4f666",
                "0808bc296f7a8db1a5ba6beea6e05909e3a53291dbfbbc717d894f78de816a4d"
              ],
              "mean": 0.833333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 815,
                "cached_tokens": 16255,
                "wall_ms": 8620
              },
              "judge_usage": {
                "input_tokens": 45252,
                "output_tokens": 1566,
                "cached_tokens": 33419,
                "wall_ms": 29003
              }
            },
            {
              "draw_index": 6,
              "generation_hash": "cb2a1d7a53ac055864c0decbf9cdef651f4309f47a3f850e68c0bb1427d14d9b",
              "status": "measured",
              "samples": [
                0.85,
                0.83,
                0.87
              ],
              "judge_sample_hashes": [
                "7d14082e506719becbaff3e47b98bcb7ac4e0c3cf151920dd0afe92dccee39fb",
                "b0ba017d435e2e9b2d86f60c22bcfc869bb629b1b068cb00e900932b9efaa910",
                "d2b13955db202c18e80038eb655b751f0ac5393217547fb9562edef9a1ec3047"
              ],
              "mean": 0.85,
              "stddev": 0.02,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 529,
                "cached_tokens": 16255,
                "wall_ms": 8570
              },
              "judge_usage": {
                "input_tokens": 45351,
                "output_tokens": 1132,
                "cached_tokens": 33485,
                "wall_ms": 24655
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.575556,
        "mean": 0.575556,
        "stddev": 0.025019,
        "samples": [
          0.58,
          0.6,
          0.55
        ],
        "generation_hash": "d66b2f6f1f47c858f0a7f6b4456e0f71a299da7de4448f45a997701c85db06a1",
        "judge_sample_hashes": [
          "6955ef2ef71a25a472cb432dcabe0582e0ff2e68a7257e3331415e65cfb653b5",
          "d945fc639c13800fe5ea12a776a568575f4e33a56d4721bb89ef106b5c5e0403",
          "d48cce683ea4c75cf3cf29cd610ae0be02f8978124774b2facef1e5945e2af98"
        ],
        "threshold": 0.7,
        "reason": "Met (a), (b), and largely (d), but left the bare `// TODO: handle expired coupons` untouched, and the gold-tier comment mostly restates what `* 0.9` does.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 45111,
          "output_tokens": 1849,
          "cached_tokens": 33325,
          "wall_ms": 34049
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.575556,
          "sd": 0.025019,
          "judge_sd_mean": 0.008389,
          "variance_ratio": 2.982358,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "6eed960c91cacd37aa5f64f8994268d158e2ce11b50e0c71e9c1bd980d81e7a2",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "237367b5415f50b0f656c1c0be73e2efab61373e5fa60765acab9a880cc366bc",
                "690aeaa7f508c3af58d74e1bacdf946e88718447d95b42fa70ce4af9489d923d",
                "26439a58b50b81ad4894a1427942080f4d75ad1a1e1b0b3b516cc1f92c5b7354"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12794,
                "output_tokens": 485,
                "cached_tokens": 12792,
                "wall_ms": 5714
              },
              "judge_usage": {
                "input_tokens": 45219,
                "output_tokens": 1231,
                "cached_tokens": 33397,
                "wall_ms": 23516
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "a6774146e03c90bf4de42a570c84cbf24e75b6bdda6b0e0db0d876e5b0697e5a",
              "status": "measured",
              "samples": [
                0.55,
                0.55,
                0.55
              ],
              "judge_sample_hashes": [
                "66edfa6fc572ffd0bc403439bc0c1e04c026839380ee45424e2401479856837c",
                "8bb0b1180ba9586d8f8bb823af4cbb4fb95f3401261557ae79383c51d152c137",
                "ac162617f0ed4de636bdc9f6e9fa94f83019f5310fa30d272800225f75f050d3"
              ],
              "mean": 0.55,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12794,
                "output_tokens": 413,
                "cached_tokens": 12792,
                "wall_ms": 6087
              },
              "judge_usage": {
                "input_tokens": 45003,
                "output_tokens": 1609,
                "cached_tokens": 33253,
                "wall_ms": 28206
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "d66b2f6f1f47c858f0a7f6b4456e0f71a299da7de4448f45a997701c85db06a1",
              "status": "measured",
              "samples": [
                0.58,
                0.6,
                0.55
              ],
              "judge_sample_hashes": [
                "6955ef2ef71a25a472cb432dcabe0582e0ff2e68a7257e3331415e65cfb653b5",
                "d945fc639c13800fe5ea12a776a568575f4e33a56d4721bb89ef106b5c5e0403",
                "d48cce683ea4c75cf3cf29cd610ae0be02f8978124774b2facef1e5945e2af98"
              ],
              "mean": 0.576667,
              "stddev": 0.025166,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12794,
                "output_tokens": 449,
                "cached_tokens": 12792,
                "wall_ms": 5346
              },
              "judge_usage": {
                "input_tokens": 45111,
                "output_tokens": 1849,
                "cached_tokens": 33325,
                "wall_ms": 34049
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.813333,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.575556,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.813333,
    "baseline_score": 0.575556,
    "delta": 0.237777,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "b47d8f54c46e68e0ea250d71f9cf3265fb5db49244365794823fac4d74a7eb52",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 7,
      "mean_input_tokens": 16257,
      "mean_output_tokens": 577.86,
      "mean_cost_usd_per_call": 0.038293,
      "median_wall_ms": 7335,
      "wall_ms_p25": 6108,
      "wall_ms_p75": 8620,
      "wall_ms_iqr": 2512
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 12794,
      "mean_output_tokens": 449,
      "mean_cost_usd_per_call": 0.030078,
      "median_wall_ms": 5714,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.008215,
    "skill_incremental_cost_usd_per_1k_calls": 8.215,
    "output_tokens_delta": 128.86,
    "median_wall_ms_delta": 1621,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.526835,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}