{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-opus-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T04:37:54.396Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-opus-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T04:37:54.396Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5-5": {
          "input_per_mtok": 4,
          "output_per_mtok": 20,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "borderline",
        "score": 0.65,
        "mean": 0.65,
        "stddev": 0.10248,
        "samples": [
          0.62,
          0.85,
          0.72
        ],
        "generation_hash": "a6768c48fe2a7dba4cbb4c9f4373ac5869159089eb3e39c86d61687c56ed2f71",
        "judge_sample_hashes": [
          "2ad92c2915b761cd282e2bc34743ffec47c1953ae03e364a4595d6487520e101",
          "9d8148ffb0cad73b2bdfdbf5db372270cea82a9f95c09ce631d701262e1fa270",
          "3d5807de1675732f45eb91031c7727af30f230f0ebd2044b1521599fc1b81321"
        ],
        "threshold": 0.7,
        "reason": "Meets (a), (b), (d) — restating comments and commented-out code removed, added comments explain mutation contract and a real trap — but the TODO remains as a rephrased note rather than implemented or removed.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 46284,
          "output_tokens": 2055,
          "cached_tokens": 34107,
          "wall_ms": 34029
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 6,
          "n_measured": 6,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.65,
          "sd": 0.10248,
          "judge_sd_mean": 0.048089,
          "variance_ratio": 2.131049,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "71ad2ea0d539659d9ad766116d1617d02d8a0cfa1d10360bedeea6fa76881644",
              "status": "measured",
              "samples": [
                0.55,
                0.45,
                0.45
              ],
              "judge_sample_hashes": [
                "bcc516da6081293e84c09db5678c33a9bd42e12608ee7fc5d2a7b1af2e698291",
                "4f0597dfc95c6e9336aef1fd67c2f6d4d722bab6513bab9002ced125e6fea569",
                "448dfebf8e209a40312413ae7fdbeea0219f457540ac35c854ad4d23ecfc821f"
              ],
              "mean": 0.483333,
              "stddev": 0.057735,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 16253,
                "output_tokens": 968,
                "cached_tokens": 540,
                "wall_ms": 12002
              },
              "judge_usage": {
                "input_tokens": 46029,
                "output_tokens": 2332,
                "cached_tokens": 33937,
                "wall_ms": 36931
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "4bd953c3299026d7859d90f812fb59f6a4b8d7ffdc31b95a19dfdf9fbdeb84d0",
              "status": "measured",
              "samples": [
                0.6,
                0.65,
                0.6
              ],
              "judge_sample_hashes": [
                "1951ff2afa8715bbf148ffff69d1d61227d50ec78c3b7621640b40ede8fb574f",
                "a1b0ca054c17ab92138fb76f30a2afe215dbd0c4917ab73e15159e2c1d72bc98",
                "a2d0455f15edd3f0326faf79db4c42898fae6dffa05935d03b622bba03662f44"
              ],
              "mean": 0.616667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 16253,
                "output_tokens": 1100,
                "cached_tokens": 16251,
                "wall_ms": 13472
              },
              "judge_usage": {
                "input_tokens": 46167,
                "output_tokens": 2198,
                "cached_tokens": 34029,
                "wall_ms": 35607
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "aece20cc30dcebcb9588d182c29f07fa8c6ab36f99d97b68d258489b79425f30",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.62
              ],
              "judge_sample_hashes": [
                "e00db546036a682e545f0564c236a44f24dccc503de11b5de35fb4098949df4d",
                "26ef4fab3d5117a0f1cc533f29ed496106309829902da96b202bd3f9ecab4330",
                "f365640759ad101e79a5f124e430baf01fc87db2b58ff498d59362edbda7cca5"
              ],
              "mean": 0.606667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 16253,
                "output_tokens": 937,
                "cached_tokens": 16251,
                "wall_ms": 11615
              },
              "judge_usage": {
                "input_tokens": 45912,
                "output_tokens": 1822,
                "cached_tokens": 33859,
                "wall_ms": 32192
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "ed8e8beae844888de5d515ead73983686e290bca5cb6cf1bbeb669776ae5a2e7",
              "status": "measured",
              "samples": [
                0.7,
                0.7,
                0.7
              ],
              "judge_sample_hashes": [
                "5a1a88b875cf23ecd2a54835659818d281e7a989f82eaee437f9f3a42c6d5385",
                "6153934c066511ed4c21635950295a85ebafc7e96c560f9118c38afd58616314",
                "0579c85f113435a98694c57e341891c9a401a713aac65faa2418e3b34ab17ae9"
              ],
              "mean": 0.7,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 16253,
                "output_tokens": 1056,
                "cached_tokens": 16251,
                "wall_ms": 13601
              },
              "judge_usage": {
                "input_tokens": 46014,
                "output_tokens": 1750,
                "cached_tokens": 33927,
                "wall_ms": 31048
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "ee5b4f3dea5c7f033d24227bce361a4da0a295fcd087b4e3f9d5ca2fa84b499b",
              "status": "measured",
              "samples": [
                0.85,
                0.72,
                0.72
              ],
              "judge_sample_hashes": [
                "d7fa3c0a4e2b19029ecaed6ae3baa4b5eec4b55434fa372f2d4cdf5a83292f38",
                "f8939a66e61024234542a23e57d0d11a4208668b03b4e0326794f8413a3d9f3f",
                "51362b0e99b327f701214c068306249df6447a311031771729db440ec173b1ad"
              ],
              "mean": 0.763333,
              "stddev": 0.075056,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 16253,
                "output_tokens": 983,
                "cached_tokens": 16251,
                "wall_ms": 12604
              },
              "judge_usage": {
                "input_tokens": 45570,
                "output_tokens": 1923,
                "cached_tokens": 33631,
                "wall_ms": 35900
              }
            },
            {
              "draw_index": 5,
              "generation_hash": "a6768c48fe2a7dba4cbb4c9f4373ac5869159089eb3e39c86d61687c56ed2f71",
              "status": "measured",
              "samples": [
                0.62,
                0.85,
                0.72
              ],
              "judge_sample_hashes": [
                "2ad92c2915b761cd282e2bc34743ffec47c1953ae03e364a4595d6487520e101",
                "9d8148ffb0cad73b2bdfdbf5db372270cea82a9f95c09ce631d701262e1fa270",
                "3d5807de1675732f45eb91031c7727af30f230f0ebd2044b1521599fc1b81321"
              ],
              "mean": 0.73,
              "stddev": 0.115326,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 16253,
                "output_tokens": 1133,
                "cached_tokens": 16251,
                "wall_ms": 13950
              },
              "judge_usage": {
                "input_tokens": 46284,
                "output_tokens": 2055,
                "cached_tokens": 34107,
                "wall_ms": 34029
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.534445,
        "mean": 0.534445,
        "stddev": 0.038921,
        "samples": [
          0.45,
          0.55,
          0.58
        ],
        "generation_hash": "d26dfd3ff4e11658f9892b3d745ff3b575e6f90dc366035a13e51507e2477346",
        "judge_sample_hashes": [
          "791ef557ecf302d34db02774fa263128615be17715800637b04a6b062f0d4fb4",
          "f139b3258ccd93adad73ac7c9b74f84db54f9b62e1cb2e4f221ad7e6eec49e8b",
          "8e3eab4d0bf32d7fcb174a05460a819a058dccff571041ed917576e914bcdfcb"
        ],
        "threshold": 0.7,
        "reason": "Meets (a) and (b) cleanly, but leaves the bare TODO verbatim (c) and its added gold-tier comment restates the 'what' (10% off) rather than the why (d).",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 45555,
          "output_tokens": 1981,
          "cached_tokens": 33621,
          "wall_ms": 35406
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.534445,
          "sd": 0.038921,
          "judge_sd_mean": 0.031078,
          "variance_ratio": 1.252365,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "6d44a0b2885b3eef90a24397b1f13553f6221a17b833cf234f5938654295b0d5",
              "status": "measured",
              "samples": [
                0.6,
                0.58,
                0.55
              ],
              "judge_sample_hashes": [
                "50e3949eaed3eb18f8a1503fc4595466c93f6190bf2bbcb226b73f30e4077547",
                "531d8b7ec6144086081e192782491f73d5d8f0235d8bca75f20192a30d533672",
                "e5ca086e693fed58599987accde608f21954e0b05bc4ad457cc81971286353a3"
              ],
              "mean": 0.576667,
              "stddev": 0.025166,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12790,
                "output_tokens": 820,
                "cached_tokens": 2144,
                "wall_ms": 10487
              },
              "judge_usage": {
                "input_tokens": 45621,
                "output_tokens": 1888,
                "cached_tokens": 33665,
                "wall_ms": 32846
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "b08bfa5fbd22db41062e4d6a58cdeb8b97eefbccbb6ab53731e57452fe17c855",
              "status": "measured",
              "samples": [
                0.5,
                0.5,
                0.5
              ],
              "judge_sample_hashes": [
                "ea79708596dec796f15075568eb221b17358bc4a51a4d9f3695a5af5af83c65a",
                "91ba3a8e0b55e4e77637277c3f34eda1194090c34b6a51feb19dfd1747053640",
                "5aaa6d19375b5e9695221452fe4c9fdcb6892007abaf528fc1691e40be261fcd"
              ],
              "mean": 0.5,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12790,
                "output_tokens": 758,
                "cached_tokens": 12788,
                "wall_ms": 9616
              },
              "judge_usage": {
                "input_tokens": 45507,
                "output_tokens": 2917,
                "cached_tokens": 33589,
                "wall_ms": 46709
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "d26dfd3ff4e11658f9892b3d745ff3b575e6f90dc366035a13e51507e2477346",
              "status": "measured",
              "samples": [
                0.45,
                0.55,
                0.58
              ],
              "judge_sample_hashes": [
                "791ef557ecf302d34db02774fa263128615be17715800637b04a6b062f0d4fb4",
                "f139b3258ccd93adad73ac7c9b74f84db54f9b62e1cb2e4f221ad7e6eec49e8b",
                "8e3eab4d0bf32d7fcb174a05460a819a058dccff571041ed917576e914bcdfcb"
              ],
              "mean": 0.526667,
              "stddev": 0.068069,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12790,
                "output_tokens": 756,
                "cached_tokens": 12788,
                "wall_ms": 9871
              },
              "judge_usage": {
                "input_tokens": 45555,
                "output_tokens": 1981,
                "cached_tokens": 33621,
                "wall_ms": 35406
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.65,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.534445,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.65,
    "baseline_score": 0.534445,
    "delta": 0.115555,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "c28455c50e6997c2f2fd5de00d89866ac15562564286dc2981bf4635aab8900d",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 6,
      "mean_input_tokens": 16253,
      "mean_output_tokens": 1029.5,
      "mean_cost_usd_per_call": 0.085602,
      "median_wall_ms": 13038,
      "wall_ms_p25": 12002,
      "wall_ms_p75": 13601,
      "wall_ms_iqr": 1599
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 12790,
      "mean_output_tokens": 778,
      "mean_cost_usd_per_call": 0.06672,
      "median_wall_ms": 9871,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.018882,
    "skill_incremental_cost_usd_per_1k_calls": 18.882,
    "output_tokens_delta": 251.5,
    "median_wall_ms_delta": 3167,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.560095,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}