{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-sonnet-5",
    "model_release_date": "2026-06-30",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T04:19:32.046Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T04:19:32.046Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "fail",
        "score": 0.602222,
        "mean": 0.602222,
        "stddev": 0.003849,
        "samples": [
          0.6,
          0.6,
          0.6
        ],
        "generation_hash": "42e52c750b2d644356e64767eaae1294e296561f1b483f129b87545a3928a5bb",
        "judge_sample_hashes": [
          "07a6d2dbd9b9417071223f814ef879ea128ee1c3cb5923da2d4cb4caadb09a01",
          "c5f5f0795f489abc6ef052db433c71b29482b33ba75c6c185a1df5a4c109bd75",
          "a198f69b6cfd38d640ba43651e83c2f082556278451a1d6d6454809db52885ff"
        ],
        "threshold": 0.7,
        "reason": "Met (a) removing all 'what' comments and (b) deleting the commented-out legacyRate line, but left the bare TODO dangling (c) and added no intent comment (d).",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44307,
          "output_tokens": 1150,
          "cached_tokens": 32789,
          "wall_ms": 21703
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.602222,
          "sd": 0.003849,
          "judge_sd_mean": 0.003849,
          "variance_ratio": 1,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "f2160e6517c260bc81b624af47cf78b93adebb00f87309b16a754c2e863f0553",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "ca193742205ab5f112e92f32007cfa30f1b4647bb6560bf24ef40bcf9a11996c",
                "cec2d325919bc2b2c907929d11926f5d9b091fcedba891916afacb5c4305fd9a",
                "9ae1a093c7fb283c2b01b6a24d0e84f68a30df85d798ec1fcd8a80355162fd31"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 597,
                "cached_tokens": 3406,
                "wall_ms": 10754
              },
              "judge_usage": {
                "input_tokens": 44433,
                "output_tokens": 1424,
                "cached_tokens": 32873,
                "wall_ms": 30926
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "4fa5af9d5b987d6fe38c6accc6e8eae4134ea1225680f16407c4f3958188eb21",
              "status": "measured",
              "samples": [
                0.62,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "7f3a75883f8b425bcf3353ecf0e0c06072b089024670c26ebf99ccf881a86f9a",
                "6810a25112a10f8259a7a62d63d4b46c07729af4f87fcf0ad45495a835393f05",
                "af06e214ef6add467f68efe28fe04293add7a86a865bbf50230a75517a68fc98"
              ],
              "mean": 0.606667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 388,
                "cached_tokens": 22673,
                "wall_ms": 19427
              },
              "judge_usage": {
                "input_tokens": 44928,
                "output_tokens": 1369,
                "cached_tokens": 33203,
                "wall_ms": 23227
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "42e52c750b2d644356e64767eaae1294e296561f1b483f129b87545a3928a5bb",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "07a6d2dbd9b9417071223f814ef879ea128ee1c3cb5923da2d4cb4caadb09a01",
                "c5f5f0795f489abc6ef052db433c71b29482b33ba75c6c185a1df5a4c109bd75",
                "a198f69b6cfd38d640ba43651e83c2f082556278451a1d6d6454809db52885ff"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 22675,
                "output_tokens": 626,
                "cached_tokens": 22673,
                "wall_ms": 10647
              },
              "judge_usage": {
                "input_tokens": 44307,
                "output_tokens": 1150,
                "cached_tokens": 32789,
                "wall_ms": 21703
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "borderline",
        "score": 0.660833,
        "mean": 0.660833,
        "stddev": 0.187446,
        "samples": [
          0.8,
          0.83,
          0.8
        ],
        "generation_hash": "0833a6e2f83052bc83c0b87d4fb906c0b4cc8e9e1f986bac55749de2d5708a3b",
        "judge_sample_hashes": [
          "5191da1add7c692c39b1b4d99aa8cf63e739b6c97f07449604d1ac4a31be33a4",
          "1897f6b1099edfe0614925843ce78afa3211441134b31229d7bd377d532827c9",
          "4f83b879c9abb08a067eb915dcf301ecf66860e973cb813326cb37adb9dbe1fd"
        ],
        "threshold": 0.7,
        "reason": "Removes all three restating comments, the commented-out legacyRate line, and the bare TODO; retained comment states the gold-tier business rule, though it borders on restating the 0.9 multiplier.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44433,
          "output_tokens": 1282,
          "cached_tokens": 32873,
          "wall_ms": 23440
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.660833,
          "sd": 0.187446,
          "judge_sd_mean": 0.024047,
          "variance_ratio": 7.794985,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "695e2c03d80f7591c1e1a9a6ebcd1959051b8c3838eb757493d777477c0251ff",
              "status": "measured",
              "samples": [
                0.65,
                0.6,
                0.7
              ],
              "judge_sample_hashes": [
                "6c8aeda91d6fa4971d7f21b7d7f2dfe0bcb3aab1ddd2396bfb54a471008270f2",
                "caf86442b9abea62cbd5e775682c108825d8f56b21701767e392505dc4bcb985",
                "59aab9dea0b90db5f753b51966ae86f400b8ba22be7bdf8e1422d8bc2ae64139"
              ],
              "mean": 0.65,
              "stddev": 0.05,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 208,
                "cached_tokens": 8503,
                "wall_ms": 6007
              },
              "judge_usage": {
                "input_tokens": 44388,
                "output_tokens": 1518,
                "cached_tokens": 32843,
                "wall_ms": 26785
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "1b8fad8695a1ed1c7601022b27f74607789481031a02fc2b4eb591c66047ca32",
              "status": "measured",
              "samples": [
                0.4,
                0.4,
                0.4
              ],
              "judge_sample_hashes": [
                "f5e55f9b67ae252f412dac7432ee7cb42179b2288017f724294a224c86d6ff34",
                "960ccc2d47a8b0fb6181e54e27c4eefbdc346e0b508ecb071cbb43b36306060b",
                "786d9c0fcbcbdfa039ac7be7e57fc4b3e88bee2d32f35b6426842002299fd766"
              ],
              "mean": 0.4,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 581,
                "cached_tokens": 19210,
                "wall_ms": 7727
              },
              "judge_usage": {
                "input_tokens": 44448,
                "output_tokens": 1873,
                "cached_tokens": 32883,
                "wall_ms": 29967
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "2320b3935b92eaabbf96e314b60cc5c264a3fc1307a42af84240c8fe8cb78743",
              "status": "measured",
              "samples": [
                0.8,
                0.75,
                0.8
              ],
              "judge_sample_hashes": [
                "8ce53f91e144ca8ffa10ee5a4d29b9fcb4d73521a87820d475d19e1497adae57",
                "37d4c18fed0ac718c4b4f8b9b029141d4d1384890f7380316c67289995af0ec1",
                "a763532cc15e6fa67fdbdd47351fa03eef383e94b0a7aef0bf6c1ef171f30872"
              ],
              "mean": 0.783333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 354,
                "cached_tokens": 19210,
                "wall_ms": 7716
              },
              "judge_usage": {
                "input_tokens": 44376,
                "output_tokens": 1547,
                "cached_tokens": 32835,
                "wall_ms": 26548
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "0833a6e2f83052bc83c0b87d4fb906c0b4cc8e9e1f986bac55749de2d5708a3b",
              "status": "measured",
              "samples": [
                0.8,
                0.83,
                0.8
              ],
              "judge_sample_hashes": [
                "5191da1add7c692c39b1b4d99aa8cf63e739b6c97f07449604d1ac4a31be33a4",
                "1897f6b1099edfe0614925843ce78afa3211441134b31229d7bd377d532827c9",
                "4f83b879c9abb08a067eb915dcf301ecf66860e973cb813326cb37adb9dbe1fd"
              ],
              "mean": 0.81,
              "stddev": 0.017321,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19212,
                "output_tokens": 223,
                "cached_tokens": 19210,
                "wall_ms": 5571
              },
              "judge_usage": {
                "input_tokens": 44433,
                "output_tokens": 1282,
                "cached_tokens": 32873,
                "wall_ms": 23440
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.602222,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.660833,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.602222,
    "baseline_score": 0.660833,
    "delta": -0.058611,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "76c9cfff2b939e172794bb9baa125a981b3edffcab6d52724db0aeb680e233a6",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 22675,
      "mean_output_tokens": 537,
      "mean_cost_usd_per_call": 0.05072,
      "median_wall_ms": 10754,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 4,
      "mean_input_tokens": 19212,
      "mean_output_tokens": 341.5,
      "mean_cost_usd_per_call": 0.041839,
      "median_wall_ms": 6861.5,
      "wall_ms_p25": 5789,
      "wall_ms_p75": 7721.5,
      "wall_ms_iqr": 1932.5
    },
    "skill_incremental_cost_usd_per_call": 0.008881,
    "skill_incremental_cost_usd_per_1k_calls": 8.881,
    "output_tokens_delta": 195.5,
    "median_wall_ms_delta": 3892.5,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.5045,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}