{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-sonnet-5",
    "model_release_date": "2026-06-30",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T06:06:43.398Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T06:06:43.398Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "borderline",
        "score": 0.641333,
        "mean": 0.641333,
        "stddev": 0.16696,
        "samples": [
          0.78,
          0.78,
          0.84
        ],
        "generation_hash": "6c4794f506da4bfe5967fa54bfe6b556dd75ecb47de24c162b4870cc6788aa34",
        "judge_sample_hashes": [
          "fd50e071e704cbf1a205bf1c83a67c8b8ac865ca1788873e3cac3c04bd92a314",
          "8c6a1398ede0fbe8f8e150727c50da41f4d98240ca2ad9b11ca9616f03172cb6",
          "f79083e48b0967f4005c34f8367912df1fbe7075ab729a96b0d8971928ade27f"
        ],
        "threshold": 0.7,
        "reason": "Meets (a), (b), (c) fully; the one added comment flags a non-obvious interaction but describes behavior/scope rather than business intent, so (d) is only partly satisfied.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44604,
          "output_tokens": 1572,
          "cached_tokens": 32987,
          "wall_ms": 34501
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 5,
          "n_measured": 5,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.641333,
          "sd": 0.16696,
          "judge_sd_mean": 0.009238,
          "variance_ratio": 18.073176,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "3b8eeadff7061728ecfb869207f7fc366cbbface6b170a4e80decc96bdb3bc35",
              "status": "measured",
              "samples": [
                0.8,
                0.8,
                0.8
              ],
              "judge_sample_hashes": [
                "569bc6a8802109de950e0baaafb78c34fc62a43bcd763d21375d18116c1cdd76",
                "70cd7b032d8832de9fa7b44352d1dc8e87ffe5246b4f2dbae78bf0474f771e99",
                "bbde6c4c3deec17dd8bde566343f84ea4d8b861203706a80a93e88cf6a91c196"
              ],
              "mean": 0.8,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 1173,
                "cached_tokens": 22673,
                "wall_ms": 19953
              },
              "judge_usage": {
                "input_tokens": 44616,
                "output_tokens": 915,
                "cached_tokens": 32995,
                "wall_ms": 19365
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "344b8c15a6fbd631f9f0d116dc3cdd20f6fa200eb74c4d479a539fcd08b7d1cd",
              "status": "measured",
              "samples": [
                0.4,
                0.4,
                0.4
              ],
              "judge_sample_hashes": [
                "f5afb684f56fdfff77b849c9751d7b371edae7645fbb29545ddff929ee391660",
                "82dbc3f1bc8da3f76283baef2f9719686a5c16e88903462de49851726c19a047",
                "326efdf6387d9f7b8f55143a883943538f26fe0fec62be174aed2d688971a3c6"
              ],
              "mean": 0.4,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 507,
                "cached_tokens": 22673,
                "wall_ms": 10362
              },
              "judge_usage": {
                "input_tokens": 44403,
                "output_tokens": 1653,
                "cached_tokens": 32853,
                "wall_ms": 26580
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "a8f36e3479c8abb15bff461dc675193bd0e127b1292870d581a151352bd44076",
              "status": "measured",
              "samples": [
                0.62,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "aff83a4468b70223fd3d53a151cf7b1d2e6dcc675628d830226a59a59b95d934",
                "5d725ff07b89ee4f9d1f477e4245bb89982fa7641df3da0734f04202ac0c90f0",
                "7ffe287bc61d95827eb0431312287f5ec135a63735465e30306e8eff293042b7"
              ],
              "mean": 0.606667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 521,
                "cached_tokens": 22673,
                "wall_ms": 7494
              },
              "judge_usage": {
                "input_tokens": 44343,
                "output_tokens": 1392,
                "cached_tokens": 32813,
                "wall_ms": 32731
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "6e81270831f757661763970de0c8ccafbbe1844dd88beb1ea34df058c31597b5",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "405d9de39b87d2d88d5294fc618d39fab0cc2061ebf0f8f06f3fe149482854aa",
                "eafd407084452af1b9628c8fc4758ec68359c07bde0f820a63e7d656b4745e8c",
                "87da419aa133f976ab18f4850241448823482d1fc9124a08c6b5fcd6ea8f0b27"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 631,
                "cached_tokens": 22673,
                "wall_ms": 8087
              },
              "judge_usage": {
                "input_tokens": 44529,
                "output_tokens": 1062,
                "cached_tokens": 32937,
                "wall_ms": 23571
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "6c4794f506da4bfe5967fa54bfe6b556dd75ecb47de24c162b4870cc6788aa34",
              "status": "measured",
              "samples": [
                0.78,
                0.78,
                0.84
              ],
              "judge_sample_hashes": [
                "fd50e071e704cbf1a205bf1c83a67c8b8ac865ca1788873e3cac3c04bd92a314",
                "8c6a1398ede0fbe8f8e150727c50da41f4d98240ca2ad9b11ca9616f03172cb6",
                "f79083e48b0967f4005c34f8367912df1fbe7075ab729a96b0d8971928ade27f"
              ],
              "mean": 0.8,
              "stddev": 0.034641,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 597,
                "cached_tokens": 22673,
                "wall_ms": 7917
              },
              "judge_usage": {
                "input_tokens": 44604,
                "output_tokens": 1572,
                "cached_tokens": 32987,
                "wall_ms": 34501
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "borderline",
        "score": 0.726667,
        "mean": 0.726667,
        "stddev": 0.11778,
        "samples": [
          0.6,
          0.6,
          0.6
        ],
        "generation_hash": "12b8a3cd53bc6d7f74716b1384d3b655e32f3677e3437531c04de079c40dbeaa",
        "judge_sample_hashes": [
          "67d8718757427fbcec518e5fe4a073b5c4847e4f9cd2e8fb07fd1167840c184a",
          "e01ac72c7e6827c5dd5979c0e45b0885a2c66c72b2256de96ad2038447344fe7",
          "be9006feaca668f0f4460f5807f8d83522f3abc08879cc85ddbe4c1cee732894"
        ],
        "threshold": 0.7,
        "reason": "Removed all 'what' restatements (a) and the commented-out legacy line (b), but left the bare TODO dangling (c) and added no intent comment for the gold-tier discount.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44607,
          "output_tokens": 1268,
          "cached_tokens": 32989,
          "wall_ms": 22645
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 5,
          "n_measured": 5,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.726667,
          "sd": 0.11778,
          "judge_sd_mean": 0.024366,
          "variance_ratio": 4.833785,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "c6fb7cfbb2fd7e7329f5dc3985a108601d1fb0f912349efb19256f41cd06bca3",
              "status": "measured",
              "samples": [
                0.84,
                0.85,
                0.84
              ],
              "judge_sample_hashes": [
                "b2da286127b160a07f23c7b15a0aac74b0f21a64034864e9a3740b1bf238cf5e",
                "ea8b4303017d1739a888604b840249333c1beb07c8ebb8ee87d4b7d33be14666",
                "35062e19b8b9a44a3b8b64b022a644305e8e7e43ca834e3818c9e9c9f56e0714"
              ],
              "mean": 0.843333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 598,
                "cached_tokens": 19210,
                "wall_ms": 8167
              },
              "judge_usage": {
                "input_tokens": 44619,
                "output_tokens": 1074,
                "cached_tokens": 32997,
                "wall_ms": 22848
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "121eb6a23ee7f45f80c8e0c4ada84d4cf3b697563258980eee67764a97dac77e",
              "status": "measured",
              "samples": [
                0.78,
                0.86,
                0.7
              ],
              "judge_sample_hashes": [
                "cbfd7b19440334126a8605efc75aedd34cbdd9197f50c2d2046f62c4aee7b1fd",
                "5bd877eaf22a1ce7dbdd960dde868edcc4945fdc479867d7e7ad908514f3f603",
                "755547c0cf513285a5cd948eaa3326a75d9f66b073c22056eb31bddf42ebf7ad"
              ],
              "mean": 0.78,
              "stddev": 0.08,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 424,
                "cached_tokens": 19210,
                "wall_ms": 6113
              },
              "judge_usage": {
                "input_tokens": 44520,
                "output_tokens": 1111,
                "cached_tokens": 32931,
                "wall_ms": 23013
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "081cb37dbec9fb7ec32679b71f12ba6d426c79ccbcf9519c5b78dcee1f550633",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "6b38ebc2c4fed338b171a2e699e0609a549acf364892b8ee36df03211c5c124c",
                "951a70263dee27136514ed2ac075d15b46626720ad131a97ff4e2182a0887565",
                "9d5456fdef32fe9c2d0fda6c9d4439a39563816f557dc364756251072217f0ad"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 358,
                "cached_tokens": 19210,
                "wall_ms": 6120
              },
              "judge_usage": {
                "input_tokens": 44340,
                "output_tokens": 1634,
                "cached_tokens": 32811,
                "wall_ms": 28080
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "020ce1dedf6523de9b17acc2e008bb86f032d39f69c984877f2a2ff26a72a2ef",
              "status": "measured",
              "samples": [
                0.8,
                0.85,
                0.78
              ],
              "judge_sample_hashes": [
                "cacccba22bd8e0a1777022648521ebd9b35ecb2a6d133169f07cd4e86d05e871",
                "8f3ff36182fbd11d0b4a7ea86ea7975e8fa486ffb149afdc49a1508e9fa39a1e",
                "2aa7892d6db86c15b1a9225fa5b33f9ee154c43eb734811e18afe1ff32fe198e"
              ],
              "mean": 0.81,
              "stddev": 0.036056,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 289,
                "cached_tokens": 19210,
                "wall_ms": 6968
              },
              "judge_usage": {
                "input_tokens": 44631,
                "output_tokens": 1292,
                "cached_tokens": 33005,
                "wall_ms": 29250
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "12b8a3cd53bc6d7f74716b1384d3b655e32f3677e3437531c04de079c40dbeaa",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "67d8718757427fbcec518e5fe4a073b5c4847e4f9cd2e8fb07fd1167840c184a",
                "e01ac72c7e6827c5dd5979c0e45b0885a2c66c72b2256de96ad2038447344fe7",
                "be9006feaca668f0f4460f5807f8d83522f3abc08879cc85ddbe4c1cee732894"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 281,
                "cached_tokens": 19210,
                "wall_ms": 5941
              },
              "judge_usage": {
                "input_tokens": 44607,
                "output_tokens": 1268,
                "cached_tokens": 32989,
                "wall_ms": 22645
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.641333,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.726667,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.641333,
    "baseline_score": 0.726667,
    "delta": -0.085334,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "8e52377012db60790fe42306356721a063e7f263bd1df916a51733ae8cd133a2",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 5,
      "mean_input_tokens": 22675,
      "mean_output_tokens": 685.8,
      "mean_cost_usd_per_call": 0.052208,
      "median_wall_ms": 8087,
      "wall_ms_p25": 7705.5,
      "wall_ms_p75": 15157.5,
      "wall_ms_iqr": 7452
    },
    "baseline": {
      "call_count": 5,
      "mean_input_tokens": 19212,
      "mean_output_tokens": 390,
      "mean_cost_usd_per_call": 0.042324,
      "median_wall_ms": 6120,
      "wall_ms_p25": 6027,
      "wall_ms_p75": 7567.5,
      "wall_ms_iqr": 1540.5
    },
    "skill_incremental_cost_usd_per_call": 0.009884,
    "skill_incremental_cost_usd_per_1k_calls": 9.884,
    "output_tokens_delta": 295.8,
    "median_wall_ms_delta": 1967,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.517055,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}