{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-haiku-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T21:20:58.469Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-5-5",
      "reported_models": [
        "claude-haiku-5-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T21:20:58.469Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-5-5": {
          "input_per_mtok": 0.1,
          "output_per_mtok": 0.5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "fail",
        "score": 0.542,
        "mean": 0.542,
        "stddev": 0.065303,
        "samples": [
          0.55,
          0.5,
          0.55
        ],
        "generation_hash": "8710a9db15d6a7a12feda8bc9166c1d19e9e264993edfc7c8d50fad7bf754418",
        "judge_sample_hashes": [
          "9ca1e4ea9c4197cf6162d8fccca046fb28cf6c9144d0e44407c6bd8f322b74f4",
          "88610180316129e1ba672f381522e09780a7ee465392d3f76838dbf9e9a933b8",
          "9bf6260413170f87b11606f95e7c1c0fb1405044076a034831b9e2070ef8fb69"
        ],
        "threshold": 0.7,
        "reason": "Met (a) and (b) cleanly, but left the bare TODO merely reworded, and the added comment leans toward describing what the line does rather than business intent.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44451,
          "output_tokens": 2045,
          "cached_tokens": 32889,
          "wall_ms": 44457
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 5,
          "n_measured": 5,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.542,
          "sd": 0.065303,
          "judge_sd_mean": 0.028127,
          "variance_ratio": 2.321719,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "245e4f1a90a385561e858826b9b0beb588096b79be5780de4a964046dcc8d551",
              "status": "measured",
              "samples": [
                0.4,
                0.45,
                0.45
              ],
              "judge_sample_hashes": [
                "95d0b83b11b2a402505d4a4ba27ff581032a1bd5ded48de069f112e2f0b618c3",
                "881216ba67755042e584a782aeef02de0612084e68b2267ddb728bee1784c0fa",
                "f115532233e840013a98153b2fae7b76815855314d790210e531ec69e628286a"
              ],
              "mean": 0.433333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 755,
                "cached_tokens": 23352,
                "wall_ms": 6257
              },
              "judge_usage": {
                "input_tokens": 44679,
                "output_tokens": 2092,
                "cached_tokens": 33041,
                "wall_ms": 36061
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "9ae70c33c25051c436db6d4d2d4ed5d2645b10ffac1452e2714ce576092d9304",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "a5f3bd0addbaa8184ead9d3c0553e532a5dc1ba277afddb7f23755d8a095ba2b",
                "500ce7cae3675d693b9511d612a7c3f124d2384fa14a5ce8d53830bbef992da4",
                "280ff94fc1792abb1e16f2ef4e94d3b6d02625940bd102067e7f67066e594adb"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 679,
                "cached_tokens": 23352,
                "wall_ms": 5721
              },
              "judge_usage": {
                "input_tokens": 44508,
                "output_tokens": 2029,
                "cached_tokens": 32927,
                "wall_ms": 34598
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "7c571ee7c08c6bd61ddc00409675beb28b0c992d3bb31339a8812f86a38a7818",
              "status": "measured",
              "samples": [
                0.5,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "7ec36f9a11930d29d9bae5fffed25188a352aca6d1fe3fb0210b3d7ebfa0a128",
                "a5d008fc77d21ee779e88b620c3f993dfc0444666be1e6d60367b9d2fb508f88",
                "60f3cd463c187a1b092f8a6c8626a4bf56585af93c453e515428001adadce650"
              ],
              "mean": 0.566667,
              "stddev": 0.057735,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 1116,
                "cached_tokens": 23352,
                "wall_ms": 7374
              },
              "judge_usage": {
                "input_tokens": 44802,
                "output_tokens": 2060,
                "cached_tokens": 33123,
                "wall_ms": 34990
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "8d4e6b7c0d15c83af5aa64de9f4c58e8302eedabc7718b8e7a4ee43d0025e34a",
              "status": "measured",
              "samples": [
                0.55,
                0.6,
                0.58
              ],
              "judge_sample_hashes": [
                "c2bcc1eca3a0b0c0e2c2e04d19a13567c87c8606089974687c4126639143d5ec",
                "2783294254d8a4289e6cd5da5b89cd6bd4a0a0438a4efdce661abb2cc5807abf",
                "0fbf878990409c57639dde8223e415423b5e4cc9e5d1e83d74bc8b2bba08d7ab"
              ],
              "mean": 0.576667,
              "stddev": 0.025166,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 1144,
                "cached_tokens": 23352,
                "wall_ms": 7338
              },
              "judge_usage": {
                "input_tokens": 44586,
                "output_tokens": 1990,
                "cached_tokens": 32979,
                "wall_ms": 37081
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "8710a9db15d6a7a12feda8bc9166c1d19e9e264993edfc7c8d50fad7bf754418",
              "status": "measured",
              "samples": [
                0.55,
                0.5,
                0.55
              ],
              "judge_sample_hashes": [
                "9ca1e4ea9c4197cf6162d8fccca046fb28cf6c9144d0e44407c6bd8f322b74f4",
                "88610180316129e1ba672f381522e09780a7ee465392d3f76838dbf9e9a933b8",
                "9bf6260413170f87b11606f95e7c1c0fb1405044076a034831b9e2070ef8fb69"
              ],
              "mean": 0.533333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 991,
                "cached_tokens": 23352,
                "wall_ms": 9752
              },
              "judge_usage": {
                "input_tokens": 44451,
                "output_tokens": 2045,
                "cached_tokens": 32889,
                "wall_ms": 44457
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.483333,
        "mean": 0.483333,
        "stddev": 0.034801,
        "samples": [
          0.5,
          0.5,
          0.57
        ],
        "generation_hash": "2530751e49bdcc4e9057d3776405a73e1417fd9a4dab1d8f07eb8cb992a59fdd",
        "judge_sample_hashes": [
          "de6251e0a773f32b28e1d97b76cd6f647f97aeb99331bb032f0da2c12c17d784",
          "19b716dee3e618add54220051515ff5c528e1315a5ce396792aac45f3a15cf21",
          "bd0331ac6b06cfeccbdef1b860d15fb8506e7f490d24788f60e6b61e660c04a7"
        ],
        "threshold": 0.7,
        "reason": "Met (a) and (b), but left the bare TODO as a mere restatement (c), and remaining comments give behavioral caveats rather than why/intent (d).",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44634,
          "output_tokens": 2751,
          "cached_tokens": 33011,
          "wall_ms": 50920
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.483333,
          "sd": 0.034801,
          "judge_sd_mean": 0.028868,
          "variance_ratio": 1.205522,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "2d78685e78620754e5d8554524f823ad05e2f68c66f2e5014f4e646d97bde747",
              "status": "measured",
              "samples": [
                0.5,
                0.45,
                0.45
              ],
              "judge_sample_hashes": [
                "dbf4d9a8d6919f53e18a6ea38b9e729c77750ccf9ecaa75fcdf99e0aad517212",
                "7e93f4b5ce825d69f89b1a44b7b6b98115a53f8dbe995dc0e8657c996a38a3cc",
                "64151b43e7237cb717b88cdee9daeefe87523e6909fd7e95fa1e70ffcf1559a9"
              ],
              "mean": 0.466667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19891,
                "output_tokens": 1338,
                "cached_tokens": 19889,
                "wall_ms": 8327
              },
              "judge_usage": {
                "input_tokens": 44763,
                "output_tokens": 2756,
                "cached_tokens": 33097,
                "wall_ms": 60614
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "9cfa4730802e9eb7f1a4c174c444179779a85b91f56d42a4d6997b8a6105b7b3",
              "status": "measured",
              "samples": [
                0.45,
                0.48,
                0.45
              ],
              "judge_sample_hashes": [
                "b36446753c3f3dc150ba97b547c0114c5a9277cd4f0b99a2ecb17ebc89e891d2",
                "539402676e14e2d32329dd47b639808646ecc98fa85b0feedb2a6006bc60aed8",
                "3a37653bb80d121eaebf398fd6ed1219a8142f936e5c0048412ab01baae56dd6"
              ],
              "mean": 0.46,
              "stddev": 0.017321,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19891,
                "output_tokens": 1012,
                "cached_tokens": 19889,
                "wall_ms": 6659
              },
              "judge_usage": {
                "input_tokens": 44475,
                "output_tokens": 2758,
                "cached_tokens": 32905,
                "wall_ms": 47165
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "2530751e49bdcc4e9057d3776405a73e1417fd9a4dab1d8f07eb8cb992a59fdd",
              "status": "measured",
              "samples": [
                0.5,
                0.5,
                0.57
              ],
              "judge_sample_hashes": [
                "de6251e0a773f32b28e1d97b76cd6f647f97aeb99331bb032f0da2c12c17d784",
                "19b716dee3e618add54220051515ff5c528e1315a5ce396792aac45f3a15cf21",
                "bd0331ac6b06cfeccbdef1b860d15fb8506e7f490d24788f60e6b61e660c04a7"
              ],
              "mean": 0.523333,
              "stddev": 0.040415,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19891,
                "output_tokens": 1204,
                "cached_tokens": 19889,
                "wall_ms": 8302
              },
              "judge_usage": {
                "input_tokens": 44634,
                "output_tokens": 2751,
                "cached_tokens": 33011,
                "wall_ms": 50920
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.542,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.483333,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.542,
    "baseline_score": 0.483333,
    "delta": 0.058667,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "aac9b8993050bd0c6c89b37734ed60645f112a6b632bbbb711668875fa8b88ec",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 5,
      "mean_input_tokens": 23354,
      "mean_output_tokens": 937,
      "mean_cost_usd_per_call": 0.002804,
      "median_wall_ms": 7338,
      "wall_ms_p25": 5989,
      "wall_ms_p75": 8563,
      "wall_ms_iqr": 2574
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 19891,
      "mean_output_tokens": 1184.67,
      "mean_cost_usd_per_call": 0.002581,
      "median_wall_ms": 8302,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.000223,
    "skill_incremental_cost_usd_per_1k_calls": 0.223,
    "output_tokens_delta": -247.67,
    "median_wall_ms_delta": -964,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.565325,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}