{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-haiku-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T20:56:04.145Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-5-5",
      "reported_models": [
        "claude-haiku-5-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T20:56:04.145Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-5-5": {
          "input_per_mtok": 0.1,
          "output_per_mtok": 0.5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "fail",
        "score": 0.5,
        "mean": 0.5,
        "stddev": 0.087401,
        "samples": [
          0.58,
          0.5,
          0.57
        ],
        "generation_hash": "ed605315c5e5800c97a80013b5d1d79f68c27d93a549faa1be6991f5f88c7c8d",
        "judge_sample_hashes": [
          "a21a62dcffa9d8e8afd1e3147b922ace6284166fb65a8a92f108912a2eaccd82",
          "0638e4967d774d89661326efa85db65699c2a047fc0ec8e5f86f1ae5551ead38",
          "f3008149f5d49371c41ad0f8a79adaf3e066c191a184afbfaa2f5de9dbd9521b"
        ],
        "threshold": 0.7,
        "reason": "Met (a) and (b) and mostly (d), but left the bare TODO dangling (c), and the mutation comment leans toward narrating behavior rather than intent.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44595,
          "output_tokens": 1885,
          "cached_tokens": 32985,
          "wall_ms": 33278
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 5,
          "n_measured": 5,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.5,
          "sd": 0.087401,
          "judge_sd_mean": 0.020265,
          "variance_ratio": 4.312904,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "3aba2681eef3fc1ebeaf0c4c70f9b04938ba978b90c7d023a70bd8cb0444d81e",
              "status": "measured",
              "samples": [
                0.4,
                0.4,
                0.45
              ],
              "judge_sample_hashes": [
                "340975ef308c8b826ff8f0eae0372cabc6f25236f7601f33cc3eda223226452b",
                "7928962adefdd1ca7d3bcaf4fd685c7f49adeb9bf97302d554dde80d42c47194",
                "7455507f26f055dddaa8cf7bc45fff9686af03293c8d350a003969f6eb5ee903"
              ],
              "mean": 0.416667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 1231,
                "cached_tokens": 23352,
                "wall_ms": 7610
              },
              "judge_usage": {
                "input_tokens": 44568,
                "output_tokens": 2745,
                "cached_tokens": 32967,
                "wall_ms": 44704
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "396aeece8c19a6097c78ba9f79c78dd069ca5085952f2eab24bf1bbbdeee327d",
              "status": "measured",
              "samples": [
                0.55,
                0.55,
                0.5
              ],
              "judge_sample_hashes": [
                "9fea34a8e5881157bbdab0375153e698d96441f42f65d196f10a7ec702946d72",
                "885a12d76c53d9c998d203158d13f71a091c789c89131de8582e753e2af6e3ac",
                "9550d29a1ab43474b1cf835a677a2d9e7ca3a784dacfc60b9b582a6a93fa547a"
              ],
              "mean": 0.533333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 1122,
                "cached_tokens": 23352,
                "wall_ms": 7206
              },
              "judge_usage": {
                "input_tokens": 44493,
                "output_tokens": 2412,
                "cached_tokens": 32917,
                "wall_ms": 106555
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "34bceedf2ca2e34d37a73518a2cd3bf9a9640c98eace02a5ddaa35fc16260f02",
              "status": "measured",
              "samples": [
                0.4,
                0.4,
                0.4
              ],
              "judge_sample_hashes": [
                "937222d78886ebc0d34bbf33aed688371131049ccffe25a11688ade18eb32081",
                "205f5caa5574a1a4f1f3c48044b366e5cbb2119587955e08ecae21c1c33b79c6",
                "f1d479b6d1935c56faa68da308b9df0d583cd6fe1263a81d757e243b701a941a"
              ],
              "mean": 0.4,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 1062,
                "cached_tokens": 23352,
                "wall_ms": 6720
              },
              "judge_usage": {
                "input_tokens": 44550,
                "output_tokens": 2222,
                "cached_tokens": 32955,
                "wall_ms": 89229
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "3be4863794a0e45dab2c04e2af9f277b7487f7120c5c1c758104db3e410e7012",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "8493ee6f882d60cf877c0ba815c574f4c1cfb39873dc6284e2178da500f15bd8",
                "dac5b42161c18ce033f81243223f6e43d792b1466b1ac14bfba1aa5117130a3d",
                "d26d70f2cb3672b3d7bed748dd4071d0f6e09a966eada176fcb650af82413353"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 1034,
                "cached_tokens": 23352,
                "wall_ms": 6742
              },
              "judge_usage": {
                "input_tokens": 44469,
                "output_tokens": 1387,
                "cached_tokens": 32901,
                "wall_ms": 32414
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "ed605315c5e5800c97a80013b5d1d79f68c27d93a549faa1be6991f5f88c7c8d",
              "status": "measured",
              "samples": [
                0.58,
                0.5,
                0.57
              ],
              "judge_sample_hashes": [
                "a21a62dcffa9d8e8afd1e3147b922ace6284166fb65a8a92f108912a2eaccd82",
                "0638e4967d774d89661326efa85db65699c2a047fc0ec8e5f86f1ae5551ead38",
                "f3008149f5d49371c41ad0f8a79adaf3e066c191a184afbfaa2f5de9dbd9521b"
              ],
              "mean": 0.55,
              "stddev": 0.043589,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 1107,
                "cached_tokens": 23352,
                "wall_ms": 6919
              },
              "judge_usage": {
                "input_tokens": 44595,
                "output_tokens": 1885,
                "cached_tokens": 32985,
                "wall_ms": 33278
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.588889,
        "mean": 0.588889,
        "stddev": 0.005092,
        "samples": [
          0.57,
          0.6,
          0.6
        ],
        "generation_hash": "90d74237331b94acf83f94a285fae8468a1bdfd364b1e2a03c465199edc8a3f8",
        "judge_sample_hashes": [
          "6223c0ed6b2b0328cea04dad08675ca541f716453716130c85081727c1a81b24",
          "7a76791704172c93e8c6c533b10b59663e5f2ff52748a84234d78513acd55a0d",
          "01e7e3b63047d735ee68c01c01c34e8de2792eced7dacf735d8a9281412c5f2f"
        ],
        "threshold": 0.7,
        "reason": "Meets (a) and (b) and keeps comments caveat/contract-focused rather than narrating code, but leaves the bare TODO dangling verbatim, violating (c) and lacking a true WHY for the gold discount.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44619,
          "output_tokens": 1997,
          "cached_tokens": 33001,
          "wall_ms": 37828
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.588889,
          "sd": 0.005092,
          "judge_sd_mean": 0.019245,
          "variance_ratio": 0.264588,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "7e2cb2b5cfa9b55aebc0817c4bd8fbae6e038f75dd876e54448e0a4b3e5e0125",
              "status": "measured",
              "samples": [
                0.55,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "21162e09fff3bc2f80ac577b84d01eeead15f809ce91927db3353626eb3b57fe",
                "78e51f419dd42cb45905d77e859d4ecde17e4b5e6caa5ee5879ba176d1f49e19",
                "8005651cd4ba7cc14fbe6375addf1d5c867ab3976e65553ca7cba80810931b64"
              ],
              "mean": 0.583333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19891,
                "output_tokens": 1060,
                "cached_tokens": 19889,
                "wall_ms": 6268
              },
              "judge_usage": {
                "input_tokens": 44727,
                "output_tokens": 1910,
                "cached_tokens": 33073,
                "wall_ms": 33359
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "42f24750c58e49f38eeea959750c3901ed5ca55fabcb2f24bb6d5c77a2060c17",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.58
              ],
              "judge_sample_hashes": [
                "3a6a0eb07e6042b85f311f2eca959db6642acf0aa6b70ca4b603e9d233b4cb03",
                "98ccf892d8d3a1e49ff82271c172a319a22b43881ae24e1c724d41eed091c31d",
                "1f269abcc05e6fecc183be865865369791bbe647a1ef731b22a9d8f6286a5c6f"
              ],
              "mean": 0.593333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19891,
                "output_tokens": 1084,
                "cached_tokens": 19889,
                "wall_ms": 6748
              },
              "judge_usage": {
                "input_tokens": 44475,
                "output_tokens": 1534,
                "cached_tokens": 32905,
                "wall_ms": 26442
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "90d74237331b94acf83f94a285fae8468a1bdfd364b1e2a03c465199edc8a3f8",
              "status": "measured",
              "samples": [
                0.57,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "6223c0ed6b2b0328cea04dad08675ca541f716453716130c85081727c1a81b24",
                "7a76791704172c93e8c6c533b10b59663e5f2ff52748a84234d78513acd55a0d",
                "01e7e3b63047d735ee68c01c01c34e8de2792eced7dacf735d8a9281412c5f2f"
              ],
              "mean": 0.59,
              "stddev": 0.017321,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19891,
                "output_tokens": 1069,
                "cached_tokens": 19889,
                "wall_ms": 6955
              },
              "judge_usage": {
                "input_tokens": 44619,
                "output_tokens": 1997,
                "cached_tokens": 33001,
                "wall_ms": 37828
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.5,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.588889,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.5,
    "baseline_score": 0.588889,
    "delta": -0.088889,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "d1829e258160c025324ddc1898470c66a0cf7fcc4c03d1d2ae9a4f49e5d889ae",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 5,
      "mean_input_tokens": 23354,
      "mean_output_tokens": 1111.2,
      "mean_cost_usd_per_call": 0.002891,
      "median_wall_ms": 6919,
      "wall_ms_p25": 6731,
      "wall_ms_p75": 7408,
      "wall_ms_iqr": 677
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 19891,
      "mean_output_tokens": 1071,
      "mean_cost_usd_per_call": 0.002525,
      "median_wall_ms": 6748,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.000366,
    "skill_incremental_cost_usd_per_1k_calls": 0.366,
    "output_tokens_delta": 40.2,
    "median_wall_ms_delta": 171,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.54312,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}