{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-sonnet-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T05:49:39.963Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T05:49:39.963Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.846667,
        "mean": 0.846667,
        "stddev": 0.01453,
        "samples": [
          0.85,
          0.85,
          0.86
        ],
        "generation_hash": "5ed1e96301dcc1ebbb4ad2af4fee4553f7e3fb83c684e33a875dad98b43e44de",
        "judge_sample_hashes": [
          "3af43dc564ac27516e02d34cde8a6a15d9c7c51b02121b7dbf58897f4d0f0a54",
          "a108e9b89c050c89062b15b7c06d69738319f0961c8f943a7f86bee02ef8ca24",
          "22213863fa79e026f52d02cd18a0fed8506aae60c7172c681f7604598e390a12"
        ],
        "threshold": 0.7,
        "reason": "Meets (a)-(d): removes all three restating comments, the commented-out legacyRate, and the bare TODO, and the one remaining comment states the gold-tier business rule.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 45405,
          "output_tokens": 998,
          "cached_tokens": 33521,
          "wall_ms": 35331
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.846667,
          "sd": 0.01453,
          "judge_sd_mean": 0.009107,
          "variance_ratio": 1.595476,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "a747bd15ad277d304ab2ae3c8bb8a047c1cac8d59cebd36ced26f2de1d03da87",
              "status": "measured",
              "samples": [
                0.87,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "186cf73127971c260a276a16e86ffbb226b3939bfed4d7a041f3306830cbcd90",
                "bc2e8d8c9903d1810921c864f57b38df31444690c2b71c3ef82de70daa3d1a26",
                "f177b191b904a1fb5b44f2ca05095666cfd4a5271038c4548b5e2872c0af1aaf"
              ],
              "mean": 0.856667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 500,
                "cached_tokens": 16255,
                "wall_ms": 5703
              },
              "judge_usage": {
                "input_tokens": 45264,
                "output_tokens": 1046,
                "cached_tokens": 33427,
                "wall_ms": 24146
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "ba1ac03166ad317f2dfbe040d923ae678792327cbd2609cddfec4f1646f32b60",
              "status": "measured",
              "samples": [
                0.82,
                0.83,
                0.84
              ],
              "judge_sample_hashes": [
                "696ab512ff2070d2622ceb6041e4e33762e5fa75a544877c85e2d5b4d8dd7429",
                "18fb285a8720661dedb785970a688923d20bc18a61ab390984213b51bb6fc665",
                "4bb94c5d69c1ade743d46fe721ee3ac9825725a09d8d6542edc3c2850ef7d60f"
              ],
              "mean": 0.83,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 469,
                "cached_tokens": 16255,
                "wall_ms": 5835
              },
              "judge_usage": {
                "input_tokens": 45171,
                "output_tokens": 1356,
                "cached_tokens": 33365,
                "wall_ms": 32115
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "5ed1e96301dcc1ebbb4ad2af4fee4553f7e3fb83c684e33a875dad98b43e44de",
              "status": "measured",
              "samples": [
                0.85,
                0.85,
                0.86
              ],
              "judge_sample_hashes": [
                "3af43dc564ac27516e02d34cde8a6a15d9c7c51b02121b7dbf58897f4d0f0a54",
                "a108e9b89c050c89062b15b7c06d69738319f0961c8f943a7f86bee02ef8ca24",
                "22213863fa79e026f52d02cd18a0fed8506aae60c7172c681f7604598e390a12"
              ],
              "mean": 0.853333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 16257,
                "output_tokens": 547,
                "cached_tokens": 16255,
                "wall_ms": 9019
              },
              "judge_usage": {
                "input_tokens": 45405,
                "output_tokens": 998,
                "cached_tokens": 33521,
                "wall_ms": 35331
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.574444,
        "mean": 0.574444,
        "stddev": 0.035953,
        "samples": [
          0.6,
          0.6,
          0.6
        ],
        "generation_hash": "70f846822d9315e2cb2a6813e5f242e9811cc3bde53ec619653d0830fbc448a8",
        "judge_sample_hashes": [
          "68ca9532de5d414e66fb51a8ff207a5511d1e535901c298ace03db8f7db1ba8a",
          "efa35757e202838508abc605e4d259ddd5615f48953c9e3c8005d319aed754bc",
          "8b6bdcfd9641ff2827ff27f444b8593ef943d4b319040d29a7923007e5c9ac9b"
        ],
        "threshold": 0.7,
        "reason": "Meets (a), (b), and largely (d), but leaves the TODO as a mere restatement rather than handling or removing it, and the loop comment partly narrates the what.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 45342,
          "output_tokens": 2031,
          "cached_tokens": 33479,
          "wall_ms": 34595
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.574444,
          "sd": 0.035953,
          "judge_sd_mean": 0.015396,
          "variance_ratio": 2.335217,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "6a8ab42cafd65385d702e806f704536de1f6b53836af495bc0981f32412a9cd9",
              "status": "measured",
              "samples": [
                0.55,
                0.55,
                0.5
              ],
              "judge_sample_hashes": [
                "a66d88a31e325730d74d644884f58d65a71fa109feb52cdd93f3f940e275fb7e",
                "3bf53b7b32fe33c373119da3d4ee9db7c720f77f8f66720a9d1452c8cbd447d2",
                "1e5dc5a7c865da0ff8f2dba57af0b13f975bff63f112ac7f71bfc0b55017b9cf"
              ],
              "mean": 0.533333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12794,
                "output_tokens": 445,
                "cached_tokens": 12792,
                "wall_ms": 5804
              },
              "judge_usage": {
                "input_tokens": 45099,
                "output_tokens": 2146,
                "cached_tokens": 33317,
                "wall_ms": 37947
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "bf99ac5d193293115dddba064611ca814288290c0687f66a052f40c9b859f637",
              "status": "measured",
              "samples": [
                0.57,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "a0ebcefe028122e5adc2ca07d49d61f6a9e749b759190f77ad35bf2212b2d01b",
                "432da07273645712dde53016acc9d7c12087a259c984386bd4f1d9c909f5b655",
                "4812ebf9c9664b9ea1275a1c717ab6249440568c51b6ccd66da7dd10fea81674"
              ],
              "mean": 0.59,
              "stddev": 0.017321,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12794,
                "output_tokens": 547,
                "cached_tokens": 12792,
                "wall_ms": 7572
              },
              "judge_usage": {
                "input_tokens": 45405,
                "output_tokens": 1910,
                "cached_tokens": 33521,
                "wall_ms": 31430
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "70f846822d9315e2cb2a6813e5f242e9811cc3bde53ec619653d0830fbc448a8",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "68ca9532de5d414e66fb51a8ff207a5511d1e535901c298ace03db8f7db1ba8a",
                "efa35757e202838508abc605e4d259ddd5615f48953c9e3c8005d319aed754bc",
                "8b6bdcfd9641ff2827ff27f444b8593ef943d4b319040d29a7923007e5c9ac9b"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12794,
                "output_tokens": 526,
                "cached_tokens": 12792,
                "wall_ms": 5882
              },
              "judge_usage": {
                "input_tokens": 45342,
                "output_tokens": 2031,
                "cached_tokens": 33479,
                "wall_ms": 34595
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.846667,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.574444,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.846667,
    "baseline_score": 0.574444,
    "delta": 0.272223,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "ed3a59bb4c749566756ac1ee66e9bb279353c38506c357b383407491d2da1e7c",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 16257,
      "mean_output_tokens": 505.33,
      "mean_cost_usd_per_call": 0.037567,
      "median_wall_ms": 5835,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 12794,
      "mean_output_tokens": 506,
      "mean_cost_usd_per_call": 0.030648,
      "median_wall_ms": 5882,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.006919,
    "skill_incremental_cost_usd_per_1k_calls": 6.919,
    "output_tokens_delta": -0.67,
    "median_wall_ms_delta": -47,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.52946,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}