{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-opus-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T05:20:20.883Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-opus-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T05:20:20.883Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5-5": {
          "input_per_mtok": 4,
          "output_per_mtok": 20,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.85,
        "mean": 0.85,
        "stddev": 0.006667,
        "samples": [
          0.85,
          0.85,
          0.85
        ],
        "generation_hash": "378eadf1b4cb90aeeb743e0bf558ca1b3543f5ee5df117dabffff271bd06c3ed",
        "judge_sample_hashes": [
          "e2b16d95ed0fbdb47018d8be586d37d93e243a768cf9b1b71823d05aed6cbbc0",
          "2839fa39cfcdd906a985e507cd1f73e97e7adb4c50dbfdb1d50d3d09956d7295",
          "ff60ce339835c4de3400fa7c5ffe5222524715c38d3adbb1665778c5ffa96e2f"
        ],
        "threshold": 0.7,
        "reason": "Removes all three restating comments, the commented-out legacy line, and the bare TODO; remaining JSDoc and inline comment explain mutation, non-idempotency, and stale-total intent rather than the what.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 45921,
          "output_tokens": 1387,
          "cached_tokens": 33865,
          "wall_ms": 25158
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.85,
          "sd": 0.006667,
          "judge_sd_mean": 0.005774,
          "variance_ratio": 1.154659,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "7997944e3aa2ad431bb743fead56115a9eaed98b9df6f212454bbfeb6bb891d8",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.87
              ],
              "judge_sample_hashes": [
                "b79ca8471738effd8f7b1c50cbe67898fe8adfc3065b565a2bdf63fe2494a605",
                "317cf5db52a1ffb383ec8135978397f610c0020459d77310107934e4e311a229",
                "a39e4435bd2ce17e19d47d51970e841112d099d56cbaeac7d4de1c5af7fd1cd3"
              ],
              "mean": 0.856667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 16253,
                "output_tokens": 1181,
                "cached_tokens": 16251,
                "wall_ms": 14261
              },
              "judge_usage": {
                "input_tokens": 46119,
                "output_tokens": 1204,
                "cached_tokens": 33997,
                "wall_ms": 28492
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "e9a1bca5254d51c40220c78ffc800646ccc5b3eefac403dbf2fa01851a1259b1",
              "status": "measured",
              "samples": [
                0.84,
                0.84,
                0.85
              ],
              "judge_sample_hashes": [
                "16c4da13e15da4d7f3505437a665dc43e27d18957b926f497d19603516244628",
                "14f32687e53d54847a979aa2b7d1a7c99bbe6ca340c351cb8e41f1b1725af426",
                "f3555b8e6c0060b9197ef2f40ee10b7874343683e6873fce482c5311ff9e06a7"
              ],
              "mean": 0.843333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 16253,
                "output_tokens": 831,
                "cached_tokens": 16251,
                "wall_ms": 10845
              },
              "judge_usage": {
                "input_tokens": 45654,
                "output_tokens": 1828,
                "cached_tokens": 33687,
                "wall_ms": 31407
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "378eadf1b4cb90aeeb743e0bf558ca1b3543f5ee5df117dabffff271bd06c3ed",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "e2b16d95ed0fbdb47018d8be586d37d93e243a768cf9b1b71823d05aed6cbbc0",
                "2839fa39cfcdd906a985e507cd1f73e97e7adb4c50dbfdb1d50d3d09956d7295",
                "ff60ce339835c4de3400fa7c5ffe5222524715c38d3adbb1665778c5ffa96e2f"
              ],
              "mean": 0.85,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 16253,
                "output_tokens": 1097,
                "cached_tokens": 16251,
                "wall_ms": 13802
              },
              "judge_usage": {
                "input_tokens": 45921,
                "output_tokens": 1387,
                "cached_tokens": 33865,
                "wall_ms": 25158
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.55,
        "mean": 0.55,
        "stddev": 0.033333,
        "samples": [
          0.55,
          0.45,
          0.55
        ],
        "generation_hash": "5cc9f3c8034972a97904af9595d8a25e5a159c3f3efba3874ac2c44b69c9fd72",
        "judge_sample_hashes": [
          "d80f8687968654e4115f0c1c2139f78a972928eed7d7c2e9e3ee1b022acfdc51",
          "795047e3cf3c18fd7707f050ec61763fda72477402a47ed84709e6876917f24b",
          "a628d1714424f036cf70a7506d0bed432d99124ed5cf07b3802b5b38a1b383b7"
        ],
        "threshold": 0.7,
        "reason": "Met (a) and (b) cleanly and added useful contract docs, but explicitly left the bare TODO (c) dangling and the retained gold-tier comment largely restates the 0.9 multiplication.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 45573,
          "output_tokens": 1722,
          "cached_tokens": 33633,
          "wall_ms": 32777
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.55,
          "sd": 0.033333,
          "judge_sd_mean": 0.045534,
          "variance_ratio": 0.732046,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "cf33d8aa2c4afbcf82851fd8cfbe56a0830d9bdd40040109f6c208da0991eaa1",
              "status": "measured",
              "samples": [
                0.55,
                0.6,
                0.5
              ],
              "judge_sample_hashes": [
                "880748bec59000454f54f09fb61fce4d15017c3853f9917095a976dd3d6a1c8f",
                "72e9e98868dc9a486f64abcde1472274fdfc9e78f98840b09db877cf72d58364",
                "d6bf1d8a995c6f50933608f814e6c1f7f7bc812a7c341f8979d980e25e5ea0ac"
              ],
              "mean": 0.55,
              "stddev": 0.05,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12790,
                "output_tokens": 751,
                "cached_tokens": 12788,
                "wall_ms": 10021
              },
              "judge_usage": {
                "input_tokens": 45639,
                "output_tokens": 2158,
                "cached_tokens": 33677,
                "wall_ms": 34519
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "d9017c92e4c475b669019da4481effa5e21b3401cb0a114f2d82bfbf113b0f9d",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.55
              ],
              "judge_sample_hashes": [
                "ee6aa91a37d2f3c47ef7896c4f19f5e670d5f5b72ccd0b4b8977ecfe3b8f738b",
                "e85378f9675ea3e12d8caec57b0b50a76ab2780bf4aebec9988890a655008280",
                "7baa6ed5d0e966302a80b44e621e726391c326a07cc5cd60a120fc22856fd0cd"
              ],
              "mean": 0.583333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12790,
                "output_tokens": 928,
                "cached_tokens": 12788,
                "wall_ms": 11462
              },
              "judge_usage": {
                "input_tokens": 45900,
                "output_tokens": 2246,
                "cached_tokens": 33851,
                "wall_ms": 36386
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "5cc9f3c8034972a97904af9595d8a25e5a159c3f3efba3874ac2c44b69c9fd72",
              "status": "measured",
              "samples": [
                0.55,
                0.45,
                0.55
              ],
              "judge_sample_hashes": [
                "d80f8687968654e4115f0c1c2139f78a972928eed7d7c2e9e3ee1b022acfdc51",
                "795047e3cf3c18fd7707f050ec61763fda72477402a47ed84709e6876917f24b",
                "a628d1714424f036cf70a7506d0bed432d99124ed5cf07b3802b5b38a1b383b7"
              ],
              "mean": 0.516667,
              "stddev": 0.057735,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12790,
                "output_tokens": 757,
                "cached_tokens": 12788,
                "wall_ms": 10070
              },
              "judge_usage": {
                "input_tokens": 45573,
                "output_tokens": 1722,
                "cached_tokens": 33633,
                "wall_ms": 32777
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.85,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.55,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.85,
    "baseline_score": 0.55,
    "delta": 0.3,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "330f82a2417cf3039439dbe0e5706b90621893352a862d89d8609f75043630cb",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 16253,
      "mean_output_tokens": 1036.33,
      "mean_cost_usd_per_call": 0.085739,
      "median_wall_ms": 13802,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 12790,
      "mean_output_tokens": 812,
      "mean_cost_usd_per_call": 0.0674,
      "median_wall_ms": 10070,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.018339,
    "skill_incremental_cost_usd_per_1k_calls": 18.339,
    "output_tokens_delta": 224.33,
    "median_wall_ms_delta": 3732,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.535195,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}