{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-haiku-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T20:25:36.917Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-5-5",
      "reported_models": [
        "claude-haiku-5-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T20:25:36.917Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-5-5": {
          "input_per_mtok": 0.1,
          "output_per_mtok": 0.5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.891111,
        "mean": 0.891111,
        "stddev": 0.011706,
        "samples": [
          0.9,
          0.9,
          0.91
        ],
        "generation_hash": "ad2712651f265507ab066a4e5c3ae3eb87d22c5cd9052b7208777f25ae0266bf",
        "judge_sample_hashes": [
          "dba7b7e3e81430c222684fca2eb8ff80ef0048c3d70cbfd485eb117964a20428",
          "d213bf731e28e940af641f8081d97c0db83c167bdf49e57e208b336600f4594b",
          "e8c480d6a724d435fc56081aebebb64c4cc07ec0f2c0ebe7d20dd6d0116b66df"
        ],
        "threshold": 0.7,
        "reason": "Explicit Critical/Required/Nit labels; secret Critical, PII logging Required, `data` name Nit; leverage-ordered with concrete fixes, though PII sits late within Required.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 47424,
          "output_tokens": 1845,
          "cached_tokens": 34871,
          "wall_ms": 91783
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.891111,
          "sd": 0.011706,
          "judge_sd_mean": 0.005258,
          "variance_ratio": 2.226322,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "0c118f4d1a232e7c3f6a9077c109b405e8228eef7f4d8167e8a502c74f79874c",
              "status": "measured",
              "samples": [
                0.88,
                0.88,
                0.88
              ],
              "judge_sample_hashes": [
                "fee56fb5de417145a11869949e6ec32f645a1b8b3dac76f072d5404f746120ac",
                "f4ab7d594950c090c8b5365337512f9fef7fb3af5d0f834a55fd2ca5978e7c8c",
                "9e3568a5c046245b2d3fd67e537aff6d52cbb1cc57e24aa08f89406c5c77d93a"
              ],
              "mean": 0.88,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 26813,
                "output_tokens": 2439,
                "cached_tokens": 3396,
                "wall_ms": 12612
              },
              "judge_usage": {
                "input_tokens": 47469,
                "output_tokens": 1562,
                "cached_tokens": 32186,
                "wall_ms": 29055
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "72043c1ea5835fefcd2f9ae2f622c84da619685a2e80653e20b6642178b19298",
              "status": "measured",
              "samples": [
                0.9,
                0.88,
                0.89
              ],
              "judge_sample_hashes": [
                "1f52ebce2ec7b7360fcc113d4054ca4734e65c2d14be383eaca21b269bd55c0b",
                "fb2f769f4a2d29d92a2675c7f7c2df4df5d4336634eb3ab69dfc41ed7c79d7cd",
                "a0bcccf4fea3fd18913a68de38c6e55f7d0bfeca5b342e4c61fa737e0487916c"
              ],
              "mean": 0.89,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 26813,
                "output_tokens": 2607,
                "cached_tokens": 26811,
                "wall_ms": 13512
              },
              "judge_usage": {
                "input_tokens": 47988,
                "output_tokens": 2160,
                "cached_tokens": 35247,
                "wall_ms": 36415
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "ad2712651f265507ab066a4e5c3ae3eb87d22c5cd9052b7208777f25ae0266bf",
              "status": "measured",
              "samples": [
                0.9,
                0.9,
                0.91
              ],
              "judge_sample_hashes": [
                "dba7b7e3e81430c222684fca2eb8ff80ef0048c3d70cbfd485eb117964a20428",
                "d213bf731e28e940af641f8081d97c0db83c167bdf49e57e208b336600f4594b",
                "e8c480d6a724d435fc56081aebebb64c4cc07ec0f2c0ebe7d20dd6d0116b66df"
              ],
              "mean": 0.903333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 26813,
                "output_tokens": 2415,
                "cached_tokens": 26811,
                "wall_ms": 13263
              },
              "judge_usage": {
                "input_tokens": 47424,
                "output_tokens": 1845,
                "cached_tokens": 34871,
                "wall_ms": 91783
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.820833,
        "mean": 0.820833,
        "stddev": 0.050653,
        "samples": [
          0.75,
          0.78,
          0.85
        ],
        "generation_hash": "890425173a2d632a23520fb2467f2319432df6c70e07d5aa6139b92fc2dccad2",
        "judge_sample_hashes": [
          "9fbcee32f417b5ab66942a68fdac1f1b68a7d2ca9f2f84e881f10307d9261493",
          "b7163fe6040904cc4fac7e28df5dab16eaf86315e54809769a50c135bb10c031",
          "ea1a15a4b7db522b0397f0ae6f7a6d705dfe7917c4a613fdaff878c0e792a68a"
        ],
        "threshold": 0.7,
        "reason": "Explicit severity labels, secret Critical, PII logging High, findings ordered by leverage with concrete fixes; but cosmetic `data` naming labeled Medium rather than Nit/Optional.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 47817,
          "output_tokens": 2474,
          "cached_tokens": 35133,
          "wall_ms": 50762
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.820833,
          "sd": 0.050653,
          "judge_sd_mean": 0.03481,
          "variance_ratio": 1.455128,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "a456ce7938a1f96d769f49fcdf60eabea41c8725fe901526fb5d698e16644421",
              "status": "measured",
              "samples": [
                0.85,
                0.7,
                0.75
              ],
              "judge_sample_hashes": [
                "fa7b09bb1def976e89e0d1aab646cbd72ec3d24951da6b2453f4e00795323488",
                "3c7b3aca338ff4450995424f973c43cc42d9aec9d0a7b20051f6c8d3071b13df",
                "0f406f34c0f9b6becc3a0b457ff5318ac0f2556368d7fae1558c32048bf3a422"
              ],
              "mean": 0.766667,
              "stddev": 0.076376,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19838,
                "output_tokens": 2278,
                "cached_tokens": 3387,
                "wall_ms": 12579
              },
              "judge_usage": {
                "input_tokens": 47685,
                "output_tokens": 2660,
                "cached_tokens": 35045,
                "wall_ms": 107246
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "4372b102814db945b8ba1ecbb854095be8be1fc3a40df7d1db1353cc6d4369c2",
              "status": "measured",
              "samples": [
                0.88,
                0.88,
                0.88
              ],
              "judge_sample_hashes": [
                "6c1c125927649515451a8919f46cf1f549d8ca4fc3eff38c0f00fa2dad56d01b",
                "b710be135fd82b70724da7cba9151143a70075e775067740293dd12b10311454",
                "ed13b7e9892e594dc0841b14eecbe8404a37a07a71ff9ba39e0aca0ee329f162"
              ],
              "mean": 0.88,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19838,
                "output_tokens": 2391,
                "cached_tokens": 19836,
                "wall_ms": 12872
              },
              "judge_usage": {
                "input_tokens": 47391,
                "output_tokens": 1846,
                "cached_tokens": 34849,
                "wall_ms": 50925
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "6fd493c516e85ae6b962ebf0af1e0661615bb62322df5c633e28bd8d802fbcfb",
              "status": "measured",
              "samples": [
                0.85,
                0.83,
                0.85
              ],
              "judge_sample_hashes": [
                "1ca9bf844f0800a6ba1b279c21a83b1f7f764cc8824c1ffaceb2092bd078aa52",
                "5fae7ebeb7e26cbf25812ce0bbcdd174b1c4dbc646cdbb585a2e587d75e553cd",
                "323e09c730ae2f19f4f2f6b62680463a9866f424ad1939e4daba3d4cc17c6a28"
              ],
              "mean": 0.843333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19838,
                "output_tokens": 2764,
                "cached_tokens": 19836,
                "wall_ms": 14074
              },
              "judge_usage": {
                "input_tokens": 47973,
                "output_tokens": 2825,
                "cached_tokens": 35237,
                "wall_ms": 44365
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "890425173a2d632a23520fb2467f2319432df6c70e07d5aa6139b92fc2dccad2",
              "status": "measured",
              "samples": [
                0.75,
                0.78,
                0.85
              ],
              "judge_sample_hashes": [
                "9fbcee32f417b5ab66942a68fdac1f1b68a7d2ca9f2f84e881f10307d9261493",
                "b7163fe6040904cc4fac7e28df5dab16eaf86315e54809769a50c135bb10c031",
                "ea1a15a4b7db522b0397f0ae6f7a6d705dfe7917c4a613fdaff878c0e792a68a"
              ],
              "mean": 0.793333,
              "stddev": 0.051316,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19838,
                "output_tokens": 2410,
                "cached_tokens": 19836,
                "wall_ms": 12863
              },
              "judge_usage": {
                "input_tokens": 47817,
                "output_tokens": 2474,
                "cached_tokens": 35133,
                "wall_ms": 50762
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.891111,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.820833,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.891111,
    "baseline_score": 0.820833,
    "delta": 0.070278,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "cb75db64ce25b96e8a29e0872c4c15e39df5216144cc494a190c78290a60df5f",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 26813,
      "mean_output_tokens": 2487,
      "mean_cost_usd_per_call": 0.003925,
      "median_wall_ms": 13263,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 4,
      "mean_input_tokens": 19838,
      "mean_output_tokens": 2460.75,
      "mean_cost_usd_per_call": 0.003214,
      "median_wall_ms": 12867.5,
      "wall_ms_p25": 12721,
      "wall_ms_p75": 13473,
      "wall_ms_iqr": 752
    },
    "skill_incremental_cost_usd_per_call": 0.000711,
    "skill_incremental_cost_usd_per_1k_calls": 0.711,
    "output_tokens_delta": 26.25,
    "median_wall_ms_delta": 395.5,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.58418,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}