{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-opus-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T04:26:04.501Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-opus-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T04:26:04.501Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5-5": {
          "input_per_mtok": 4,
          "output_per_mtok": 20,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.906667,
        "mean": 0.906667,
        "stddev": 0.01453,
        "samples": [
          0.93,
          0.92,
          0.9
        ],
        "generation_hash": "0613c112846c0ebcacd61cce70080b3704497532f44b03b411a9d5e95c478f2d",
        "judge_sample_hashes": [
          "e7905cfafed9517c7c0a8364dd4dafd342425f6a8bcab784e9cad858a3e9d3b6",
          "a466e1377d2295ac2545934860ce8ad7de2c271eefbbb69a90ea1031abcb8dbd",
          "c3dd99ca0317c9ea4fd9a3ed75722a805934a104fda5f15f08fb5e53585a23f8"
        ],
        "threshold": 0.7,
        "reason": "Explicit Critical/Required/Nit labels; secret Critical, PII Required, `data` Nit; ordered by leverage with concrete fixes each, only the unlabeled 'Verification gap' section slightly imperfect.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 48690,
          "output_tokens": 1584,
          "cached_tokens": 35711,
          "wall_ms": 29160
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.906667,
          "sd": 0.01453,
          "judge_sd_mean": 0.012274,
          "variance_ratio": 1.183803,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "f83b1d7c177a183c991220b8068ce75b3a0d269230c9249a0e25f2f6e28bdd95",
              "status": "measured",
              "samples": [
                0.9,
                0.89,
                0.88
              ],
              "judge_sample_hashes": [
                "e9b334ea8fff5cede9baf965cdfb24671672a166b6798c57e9ea7e680651f0ed",
                "171407964f66aff6ff85233d998327a137da0f9698c497ec8809833aff057840",
                "5bd6f776717278c9925a2553b799182cfa6ec3134a420f5d9e9615aa7d828262"
              ],
              "mean": 0.89,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 19712,
                "output_tokens": 1708,
                "cached_tokens": 540,
                "wall_ms": 23160
              },
              "judge_usage": {
                "input_tokens": 48912,
                "output_tokens": 2169,
                "cached_tokens": 35859,
                "wall_ms": 34807
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "704894302a7a4fccc49a87e986d5cc4fcff3b63a654492718fd196e9666027c0",
              "status": "measured",
              "samples": [
                0.9,
                0.92,
                0.92
              ],
              "judge_sample_hashes": [
                "000207d461d8ae8f02af21aee03510af41d26ec8e69d378aeadc637710854e55",
                "34d79bc0b09931c32b5e888a6d2cb75a7c4549133f4771d7b6e32e6c580c337e",
                "9d2a5584ae3ac4bf4c86f0c6fc884284be67daac584bdf4f0cd1eb7528c85b87"
              ],
              "mean": 0.913333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 19712,
                "output_tokens": 2059,
                "cached_tokens": 19710,
                "wall_ms": 22052
              },
              "judge_usage": {
                "input_tokens": 48546,
                "output_tokens": 1445,
                "cached_tokens": 35615,
                "wall_ms": 26612
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "0613c112846c0ebcacd61cce70080b3704497532f44b03b411a9d5e95c478f2d",
              "status": "measured",
              "samples": [
                0.93,
                0.92,
                0.9
              ],
              "judge_sample_hashes": [
                "e7905cfafed9517c7c0a8364dd4dafd342425f6a8bcab784e9cad858a3e9d3b6",
                "a466e1377d2295ac2545934860ce8ad7de2c271eefbbb69a90ea1031abcb8dbd",
                "c3dd99ca0317c9ea4fd9a3ed75722a805934a104fda5f15f08fb5e53585a23f8"
              ],
              "mean": 0.916667,
              "stddev": 0.015275,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 19712,
                "output_tokens": 2121,
                "cached_tokens": 19710,
                "wall_ms": 22158
              },
              "judge_usage": {
                "input_tokens": 48690,
                "output_tokens": 1584,
                "cached_tokens": 35711,
                "wall_ms": 29160
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.830833,
        "mean": 0.830833,
        "stddev": 0.065737,
        "samples": [
          0.87,
          0.88,
          0.88
        ],
        "generation_hash": "e94e52271546de520ff49c5af7da105523171f52de052c70680c5244bb506e4c",
        "judge_sample_hashes": [
          "d55e37048bd70d8933ed61c941e8e11e9f94e2026a7be3d397e538bd19a6475b",
          "21b020c292354bb50b2c7e10697014ff18c0acc9f69921f9ea962aeafdf7980d",
          "ddec0368a3764ecc83471c86eb801b9ad9bd9cf4f520fbb396338a595a7e9e88"
        ],
        "threshold": 0.7,
        "reason": "Every finding carries an explicit severity; secret is Critical, PII logging High, naming Low; ordered by leverage with concrete fixes, but labels deviate from the skill's taxonomy and `data` sits at Medium.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 48144,
          "output_tokens": 2219,
          "cached_tokens": 35347,
          "wall_ms": 34369
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.830833,
          "sd": 0.065737,
          "judge_sd_mean": 0.036845,
          "variance_ratio": 1.78415,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "3bb7380edfeb54b9227bf9164daef9884287bf385a6c2427972d6921b8fef682",
              "status": "measured",
              "samples": [
                0.85,
                0.86,
                0.85
              ],
              "judge_sample_hashes": [
                "b4c1fd969eb492c5b6a1f53e06393d0ad154dad3728f6b775b9988489517c354",
                "d54d62e33e186c0f1d09b7c5d3c3d1989e49ae501a9a322011f95a39cf3d6841",
                "f5d50c846e7a2656590758cb5b0e1453375cd33dc816d011e554d030276ebe0a"
              ],
              "mean": 0.853333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12737,
                "output_tokens": 1535,
                "cached_tokens": 531,
                "wall_ms": 17132
              },
              "judge_usage": {
                "input_tokens": 48462,
                "output_tokens": 2844,
                "cached_tokens": 35559,
                "wall_ms": 43488
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "571bbb71e504c59eba578ac70219220fb40c3244204228b9c5f305ebba4a466a",
              "status": "measured",
              "samples": [
                0.87,
                0.85,
                0.86
              ],
              "judge_sample_hashes": [
                "12f651095f313095b5626456bb4f13b905eafc05500fc70bae8cd28729fd0703",
                "77b2bc05602e94676f490877f9fd28c225d7e99a86ffc66624606c65331d0557",
                "fc73f608666bc66baf9b478eef14bf4ca8df6d40f9f7eb6ab0414810622a5beb"
              ],
              "mean": 0.86,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12737,
                "output_tokens": 1381,
                "cached_tokens": 12735,
                "wall_ms": 15638
              },
              "judge_usage": {
                "input_tokens": 48000,
                "output_tokens": 2439,
                "cached_tokens": 35251,
                "wall_ms": 48791
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "689f3d6143e5bbce785723c38b3fd8e9511dc3a88acfb0322cb265c738c23046",
              "status": "measured",
              "samples": [
                0.6,
                0.75,
                0.85
              ],
              "judge_sample_hashes": [
                "d3fb77cfd7f7d7925aa43826e06a0d4db6971a6ac88dc5f93d48fb137b34c30e",
                "e376bb7d770390b4a4f8e25e36abf59e3761d577ec0a4e00065917ffe4500ee7",
                "c3ee03fbc8296122a63bc6d9825fb1afb1fab223b367dc0b8fbc493cc49ed719"
              ],
              "mean": 0.733333,
              "stddev": 0.125831,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12737,
                "output_tokens": 1518,
                "cached_tokens": 12735,
                "wall_ms": 16314
              },
              "judge_usage": {
                "input_tokens": 48369,
                "output_tokens": 2708,
                "cached_tokens": 35497,
                "wall_ms": 42710
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "e94e52271546de520ff49c5af7da105523171f52de052c70680c5244bb506e4c",
              "status": "measured",
              "samples": [
                0.87,
                0.88,
                0.88
              ],
              "judge_sample_hashes": [
                "d55e37048bd70d8933ed61c941e8e11e9f94e2026a7be3d397e538bd19a6475b",
                "21b020c292354bb50b2c7e10697014ff18c0acc9f69921f9ea962aeafdf7980d",
                "ddec0368a3764ecc83471c86eb801b9ad9bd9cf4f520fbb396338a595a7e9e88"
              ],
              "mean": 0.876667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12737,
                "output_tokens": 1429,
                "cached_tokens": 12735,
                "wall_ms": 15870
              },
              "judge_usage": {
                "input_tokens": 48144,
                "output_tokens": 2219,
                "cached_tokens": 35347,
                "wall_ms": 34369
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.906667,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.830833,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.906667,
    "baseline_score": 0.830833,
    "delta": 0.075834,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "6eec133f88fec12cf0d3bc264289d85559794dc84509b62318fd6cf67bdeca21",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 19712,
      "mean_output_tokens": 1962.67,
      "mean_cost_usd_per_call": 0.118101,
      "median_wall_ms": 22158,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 4,
      "mean_input_tokens": 12737,
      "mean_output_tokens": 1465.75,
      "mean_cost_usd_per_call": 0.080263,
      "median_wall_ms": 16092,
      "wall_ms_p25": 15754,
      "wall_ms_p75": 16723,
      "wall_ms_iqr": 969
    },
    "skill_incremental_cost_usd_per_call": 0.037838,
    "skill_incremental_cost_usd_per_1k_calls": 37.838,
    "output_tokens_delta": 496.92,
    "median_wall_ms_delta": 6066,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.579245,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}