{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "documentation-and-adrs",
    "version": "0.0.0",
    "content_hash": "b67a9f07ed1057963d3d8ab04bb070ee74d93174eee48b563aa235057005c158",
    "tokens": 2454
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "6cce54e7d9d0fc923eda5499c75fcc68974b3d8714b2a726e2ad1d77e4b5acc0",
    "case_count": 1,
    "canary": "86e8fbf6-9c40-ec04-dacc-03e730017455"
  },
  "run": {
    "model_id": "claude-haiku-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T20:37:47.549Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-5-5",
      "reported_models": [
        "claude-haiku-5-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T20:37:47.549Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-5-5": {
          "input_per_mtok": 0.1,
          "output_per_mtok": 0.5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "comment-intent-not-implementation",
        "mode": "with_skill",
        "outcome": "fail",
        "score": 0.596667,
        "mean": 0.596667,
        "stddev": 0.005774,
        "samples": [
          0.6,
          0.6,
          0.57
        ],
        "generation_hash": "5833fbc71b1a6108f37cc8b79de31e9be568c03998b09fd80978ec1631467b9f",
        "judge_sample_hashes": [
          "73f0c9918657eac31c3bae1925f7bd21aa86eaa3834ea646d51fb5720141e1d7",
          "4e105edc2f5e2eb967b7b7504b8611d77944e056df3968751e0242cdf5b739d9",
          "70c294e2aba5f034355761920a02d1342b9730f962771f03a9b2b4fbd81fb42c"
        ],
        "threshold": 0.7,
        "reason": "Met (a), (b), and (d) with useful contract-level header comments, but left the TODO as a mere restatement rather than implementing or removing it.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44709,
          "output_tokens": 1785,
          "cached_tokens": 33061,
          "wall_ms": 34987
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.596667,
          "sd": 0.005774,
          "judge_sd_mean": 0.005774,
          "variance_ratio": 1,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "0b045c93caba0e50b0f026d1bd0f27850935ee419e5a9a9f6f6a5dd8a55fa8c1",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "6c336083638564184e01aeb4c81b47383c669202798e05656578212d29c367ef",
                "1a1064c98a86889f61544444e6c39f130d360c65587b333802d1ec5bd7bb348f",
                "0fdba6b0fda0dec46c336725e15fdbac7a7c742f6645aee70804c4172a7f06b2"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 1251,
                "cached_tokens": 3396,
                "wall_ms": 8153
              },
              "judge_usage": {
                "input_tokens": 44847,
                "output_tokens": 1804,
                "cached_tokens": 33153,
                "wall_ms": 33482
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "74681236c91d3039d2730d5f0d9a70415d4af5c2d23374da40a15fb068e8da03",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "d1517716f74343781af663665e2e961ba4875d3983ac6aba6877d3baf4a194f0",
                "4678c8ff151b2f93544ffdfe801dd57c52c2d3226774b646bbc103bfd564cd20",
                "804690a9e8c3cfb8a6d97015503f81209d7dd647cf93741a9b71f0e7a307fb7e"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 1195,
                "cached_tokens": 23352,
                "wall_ms": 7635
              },
              "judge_usage": {
                "input_tokens": 44628,
                "output_tokens": 2032,
                "cached_tokens": 33007,
                "wall_ms": 40611
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "5833fbc71b1a6108f37cc8b79de31e9be568c03998b09fd80978ec1631467b9f",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.57
              ],
              "judge_sample_hashes": [
                "73f0c9918657eac31c3bae1925f7bd21aa86eaa3834ea646d51fb5720141e1d7",
                "4e105edc2f5e2eb967b7b7504b8611d77944e056df3968751e0242cdf5b739d9",
                "70c294e2aba5f034355761920a02d1342b9730f962771f03a9b2b4fbd81fb42c"
              ],
              "mean": 0.59,
              "stddev": 0.017321,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 23354,
                "output_tokens": 965,
                "cached_tokens": 23352,
                "wall_ms": 6687
              },
              "judge_usage": {
                "input_tokens": 44709,
                "output_tokens": 1785,
                "cached_tokens": 33061,
                "wall_ms": 34987
              }
            }
          ]
        }
      },
      {
        "id": "comment-intent-not-implementation",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.561111,
        "mean": 0.561111,
        "stddev": 0.041943,
        "samples": [
          0.55,
          0.55,
          0.6
        ],
        "generation_hash": "6acdb414ec712d8a9c1d65bc17208e3ba690651252fe780b45bf2b0698376b7d",
        "judge_sample_hashes": [
          "6b93f74bc67c5a0fc36c06cd57f53d4b61e5e4d117b7ecc6e3000e5de615de1a",
          "d522e223a48af83c7b3c5e5b0486f708e1806a83bf2c8f1485b961c464d14b6a",
          "fe1a6c243ad79ad1c7e897bb2decab8bed9fd4edd0181563f8a762b3d051ac0d"
        ],
        "threshold": 0.7,
        "reason": "Met (a) and (b) and largely (d) with useful mutation/precondition notes, but left the TODO as a mere restatement rather than resolving or removing it.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "b38f01e6671fc0b28a94de550fe70f879134a3168b35e5b1fb53a9a6773478b0"
        },
        "judge_usage": {
          "input_tokens": 44649,
          "output_tokens": 1867,
          "cached_tokens": 33021,
          "wall_ms": 143872
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.561111,
          "sd": 0.041943,
          "judge_sd_mean": 0.028868,
          "variance_ratio": 1.452924,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "37556fd8badc7463f3d4e53c45b9daf0bb3bb989c1c684717f13a198486053db",
              "status": "measured",
              "samples": [
                0.55,
                0.55,
                0.45
              ],
              "judge_sample_hashes": [
                "78ee7e8bee5aaaa5bd00a207a2f561174be3dfaa12c682d7dcd06fadfae325d7",
                "063f3011c5da0ba309dd0c9ea9c8ba0278f593ab8e1d826692f13bcc88c636c3",
                "8cce4b8d3dab0d39bda7135af8a72e33d263e692102225e432ad37b7ccd2c1ac"
              ],
              "mean": 0.516667,
              "stddev": 0.057735,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19891,
                "output_tokens": 1142,
                "cached_tokens": 9273,
                "wall_ms": 7483
              },
              "judge_usage": {
                "input_tokens": 44550,
                "output_tokens": 2797,
                "cached_tokens": 32955,
                "wall_ms": 168270
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "efb79aac1c049e3dc63230b29fff9b1bc09d24fbdfcee27147532af9ed0f535c",
              "status": "measured",
              "samples": [
                0.6,
                0.6,
                0.6
              ],
              "judge_sample_hashes": [
                "9cc05c7ba9682788e58e3319ae382df8a5ae08bd6418423e5266dd974578554d",
                "38db85e8608c6b1fa47a52e7289b3453d19415b94998153075d727505033c146",
                "6caf167ce4951c307d348d9a4a4f1bbd3ac9afa1d56275cd8fdda6ce418d566a"
              ],
              "mean": 0.6,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19891,
                "output_tokens": 1224,
                "cached_tokens": 19889,
                "wall_ms": 7637
              },
              "judge_usage": {
                "input_tokens": 44622,
                "output_tokens": 1413,
                "cached_tokens": 33003,
                "wall_ms": 29483
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "6acdb414ec712d8a9c1d65bc17208e3ba690651252fe780b45bf2b0698376b7d",
              "status": "measured",
              "samples": [
                0.55,
                0.55,
                0.6
              ],
              "judge_sample_hashes": [
                "6b93f74bc67c5a0fc36c06cd57f53d4b61e5e4d117b7ecc6e3000e5de615de1a",
                "d522e223a48af83c7b3c5e5b0486f708e1806a83bf2c8f1485b961c464d14b6a",
                "fe1a6c243ad79ad1c7e897bb2decab8bed9fd4edd0181563f8a762b3d051ac0d"
              ],
              "mean": 0.566667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19891,
                "output_tokens": 1302,
                "cached_tokens": 19889,
                "wall_ms": 7665
              },
              "judge_usage": {
                "input_tokens": 44649,
                "output_tokens": 1867,
                "cached_tokens": 33021,
                "wall_ms": 143872
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.596667,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.561111,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.596667,
    "baseline_score": 0.561111,
    "delta": 0.035556,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "ace4c1d1a699d25b67fe54f3f8fb39fe8beeecba2d46b380ea35f30babe46e8b",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 23354,
      "mean_output_tokens": 1137,
      "mean_cost_usd_per_call": 0.002904,
      "median_wall_ms": 7635,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 19891,
      "mean_output_tokens": 1222.67,
      "mean_cost_usd_per_call": 0.0026,
      "median_wall_ms": 7637,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.000304,
    "skill_incremental_cost_usd_per_1k_calls": 0.304,
    "output_tokens_delta": -85.67,
    "median_wall_ms_delta": -2,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.53809,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}