{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-sonnet-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T04:06:28.614Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T04:06:28.614Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "borderline",
        "score": 0.780555,
        "mean": 0.780555,
        "stddev": 0.140529,
        "samples": [
          0.87,
          0.87,
          0.85
        ],
        "generation_hash": "f08e5bec72b706d29e4e528f47a2746ba560d45075f5523bebd3d4af8e6f3218",
        "judge_sample_hashes": [
          "c9713c9782b166fbfe0edf61e4c35ff029998cb914b8dcd9a684bd6f0af53fef",
          "0a6f13890c69d263ad41c0510df72d1ea8bdcd6962da3ac850fedf240bf94ca9",
          "a499859bfcd3934c4b9c4d29de14b7e0cc133b0b0ec8ad6531fad751193970ef"
        ],
        "threshold": 0.7,
        "reason": "Meets (a)-(d): removed all three restating comments, the commented-out legacyRate, and the bare TODO; added one intent comment explaining gold-tier 10% off; exemplary via flagged ambiguities.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 45495,
          "output_tokens": 822,
          "cached_tokens": 33581,
          "wall_ms": 22270
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 6,
          "n_measured": 6,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.780555,
          "sd": 0.140529,
          "judge_sd_mean": 0.03289,
          "variance_ratio": 4.272697,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "2c2b622374d7d96a62492fa5ca30bd9f672adfd458e82636b481bec8ea46e707",
              "status": "measured",
              "samples": [
                0.5,
                0.55,
                0.45
              ],
              "judge_sample_hashes": [
                "4a1c565a989a4dfbe3c7a06d735341f05665629bcecf22d9c60b0cfda7f2e1c3",
                "c7e34e94d55b18ba93ab0b0ad2dce82c8f360da7bc7b24e720e9c32da008578d",
                "e79be84c15fb1d1f3b0436c627f60a88aad50229ffb993910300ba692aa17fba"
              ],
              "mean": 0.5,
              "stddev": 0.05,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 946,
                "cached_tokens": 540,
                "wall_ms": 9625
              },
              "judge_usage": {
                "input_tokens": 45666,
                "output_tokens": 2851,
                "cached_tokens": 33695,
                "wall_ms": 47380
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "048d2d7316a753230f6b791ec460d7efdc388569b183e705905c7feea1b5fad1",
              "status": "measured",
              "samples": [
                0.85,
                0.8,
                0.7
              ],
              "judge_sample_hashes": [
                "3cb7334d98f9df054dd91ce59692ffda3830366c1fc3de589ae8264faefc0e3a",
                "d13a2056d6fba14d9067caf108a415db958b9761c15acb6a16509914a83bd908",
                "637b6fcd108e8100857fb04fe83aba93a77a58329bd68b7f5297bbfe2f4db5ae"
              ],
              "mean": 0.783333,
              "stddev": 0.076376,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 515,
                "cached_tokens": 16255,
                "wall_ms": 6266
              },
              "judge_usage": {
                "input_tokens": 45309,
                "output_tokens": 1325,
                "cached_tokens": 33457,
                "wall_ms": 31677
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "9be0a801683e707ef61d7f3c0c912932e8972daea454683c2f94f995f27a0961",
              "status": "measured",
              "samples": [
                0.85,
                0.82,
                0.85
              ],
              "judge_sample_hashes": [
                "e44be4791d01428bad567753570e7a3e0dd42c3842380e34444de833454f353e",
                "db580faa09de78cb2a54cf03f9a3308c177ac772291ad18a9b92c2484c3a77fa",
                "e8947d74a28c817c8cc5f6ba5007fdcd5f2792d1a1adf40ac66a72f9e64d03fe"
              ],
              "mean": 0.84,
              "stddev": 0.017321,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 546,
                "cached_tokens": 16255,
                "wall_ms": 10827
              },
              "judge_usage": {
                "input_tokens": 45402,
                "output_tokens": 1212,
                "cached_tokens": 33519,
                "wall_ms": 31608
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "488649cb5aa2fca20ccc6484757a9c4fef0e38844b3c4321e69328da5fbabe50",
              "status": "measured",
              "samples": [
                0.86,
                0.8,
                0.84
              ],
              "judge_sample_hashes": [
                "909f0a99497834b8848c4e7e04b3e93e6a3983d6f828f0d851ee4a99ad25d798",
                "a0985760730a13dfc25d31d5acc4e35624b4a1255334b26147f83ca9db3e728e",
                "d94d5c37c12958bdcc9969edda839325f17d1aafac3560e4313a6e06c7221322"
              ],
              "mean": 0.833333,
              "stddev": 0.030551,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 432,
                "cached_tokens": 16255,
                "wall_ms": 6424
              },
              "judge_usage": {
                "input_tokens": 45060,
                "output_tokens": 1294,
                "cached_tokens": 33291,
                "wall_ms": 24266
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "1c2322c24efc366fbb1e3908ad6ba457924c825129c3ffb537b26355ded544f9",
              "status": "measured",
              "samples": [
                0.87,
                0.85,
                0.87
              ],
              "judge_sample_hashes": [
                "6e255f1d32053f13377691a8d7c041adf3cb7199018d8b7bef3218994c90177b",
                "2a1fcb718c4a26f3493cd30419f8c726fce8237ab0a4fc335369311dbdd342f5",
                "d30245b263055eb8e92f6227fa8d8e431a890a49dafa8bad69d73a4265f61dfd"
              ],
              "mean": 0.863333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 538,
                "cached_tokens": 16255,
                "wall_ms": 7123
              },
              "judge_usage": {
                "input_tokens": 45378,
                "output_tokens": 960,
                "cached_tokens": 33503,
                "wall_ms": 21670
              }
            },
            {
              "draw_index": 5,
              "generation_hash": "f08e5bec72b706d29e4e528f47a2746ba560d45075f5523bebd3d4af8e6f3218",
              "status": "measured",
              "samples": [
                0.87,
                0.87,
                0.85
              ],
              "judge_sample_hashes": [
                "c9713c9782b166fbfe0edf61e4c35ff029998cb914b8dcd9a684bd6f0af53fef",
                "0a6f13890c69d263ad41c0510df72d1ea8bdcd6962da3ac850fedf240bf94ca9",
                "a499859bfcd3934c4b9c4d29de14b7e0cc133b0b0ec8ad6531fad751193970ef"
              ],
              "mean": 0.863333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 577,
                "cached_tokens": 16255,
                "wall_ms": 7013
              },
              "judge_usage": {
                "input_tokens": 45495,
                "output_tokens": 822,
                "cached_tokens": 33581,
                "wall_ms": 22270
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.572222,
        "mean": 0.572222,
        "stddev": 0.025459,
        "samples": [
          0.55,
          0.5,
          0.6
        ],
        "generation_hash": "edea729eb02e5a0b97aebce7d906ede5b06617a48f12224f3945ab0996742e9c",
        "judge_sample_hashes": [
          "73852b009fb701b631a13ea6f5e647bfa970d7a46ffe26dba36daa90afc655e7",
          "d599272be019ce682424861cd98c1323144cad8e24db965f56d4d066fa974cb0",
          "a48d1f0ca5ad21222720305a04d6e19edf6e7f16d23f303da011761d97dbb0bb"
        ],
        "threshold": 0.7,
        "reason": "Met (a),(b),(d): removed restating comments, deleted commented-out legacy line, added intent comment for gold tier; missed (c) by keeping a merely reworded TODO.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44967,
          "output_tokens": 2577,
          "cached_tokens": 33229,
          "wall_ms": 41691
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.572222,
          "sd": 0.025459,
          "judge_sd_mean": 0.026289,
          "variance_ratio": 0.968428,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "d7c642f350ea992904dd470348ac71d85a3049a90d0a9fd03001398a6a50316f",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "9e42f0f25bfb2c489223403e70a93d425f1e01dfe9556e7a611fd69963d28751",
                "86a6b47f021e87ad874233fcc3fe9e799078cfce177926721c79e59eb520d053",
                "849bd280e971e4e2724289fac73b269ae842d76aa8ed8b38887ae24801972dd0"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12794,
                "output_tokens": 471,
                "cached_tokens": 2144,
                "wall_ms": 6048
              },
              "judge_usage": {
                "input_tokens": 45177,
                "output_tokens": 1721,
                "cached_tokens": 33369,
                "wall_ms": 38121
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "c8d50171adfe0226500abfef13d835770be422f26abdad41502acfbf97bb8371",
              "status": "measured",
              "samples": [
                0.55,
                0.6,
                0.55
              ],
              "judge_sample_hashes": [
                "ced6af2f7b98144a41bb3cf68b37b3f21b874d63d043b8846f38d06cd54312c5",
                "7d65f43b764cec52ed3ee84379b0141f0b0efa80e69a4afce2946a07be23baaa",
                "63d361b82c7bda76b698b0049769921924accc70c2e4d0b9ed1b37f9c6a35604"
              ],
              "mean": 0.566667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12794,
                "output_tokens": 394,
                "cached_tokens": 12792,
                "wall_ms": 5040
              },
              "judge_usage": {
                "input_tokens": 44946,
                "output_tokens": 2040,
                "cached_tokens": 33215,
                "wall_ms": 33357
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "edea729eb02e5a0b97aebce7d906ede5b06617a48f12224f3945ab0996742e9c",
              "status": "measured",
              "samples": [
                0.55,
                0.5,
                0.6
              ],
              "judge_sample_hashes": [
                "73852b009fb701b631a13ea6f5e647bfa970d7a46ffe26dba36daa90afc655e7",
                "d599272be019ce682424861cd98c1323144cad8e24db965f56d4d066fa974cb0",
                "a48d1f0ca5ad21222720305a04d6e19edf6e7f16d23f303da011761d97dbb0bb"
              ],
              "mean": 0.55,
              "stddev": 0.05,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12794,
                "output_tokens": 401,
                "cached_tokens": 12792,
                "wall_ms": 5391
              },
              "judge_usage": {
                "input_tokens": 44967,
                "output_tokens": 2577,
                "cached_tokens": 33229,
                "wall_ms": 41691
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.780555,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.572222,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.780555,
    "baseline_score": 0.572222,
    "delta": 0.208333,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "47fc7a4455f21f1f848f3b4968cfbc744ed5791cf363f1c64a4ab9228444b20c",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 6,
      "mean_input_tokens": 16257,
      "mean_output_tokens": 592.33,
      "mean_cost_usd_per_call": 0.038437,
      "median_wall_ms": 7068,
      "wall_ms_p25": 6424,
      "wall_ms_p75": 9625,
      "wall_ms_iqr": 3201
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 12794,
      "mean_output_tokens": 422,
      "mean_cost_usd_per_call": 0.029808,
      "median_wall_ms": 5391,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.008629,
    "skill_incremental_cost_usd_per_1k_calls": 8.629,
    "output_tokens_delta": 170.33,
    "median_wall_ms_delta": 1677,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.537285,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}