{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-haiku-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T20:44:58.990Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-5-5",
      "reported_models": [
        "claude-haiku-5-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T20:44:58.990Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-5-5": {
          "input_per_mtok": 0.1,
          "output_per_mtok": 0.5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.860833,
        "mean": 0.860833,
        "stddev": 0.068388,
        "samples": [
          0.92,
          0.92,
          0.92
        ],
        "generation_hash": "250b0a366f5f45f355c5d778fd86d00ec8a3d49d271634da2b2b039189136ebb",
        "judge_sample_hashes": [
          "e97123bce6e0dbdf18ad7c9afe3f4bbf7fee7e04f8fb483f1bec8bea9ff7f75b",
          "808872493befc82e61264e9a389414b0a33b6b30746d33872282f55bdb31c396",
          "47b624b95449595be4a76cee7917fbb537cfdb64872b83985abcca98c43998b2"
        ],
        "threshold": 0.7,
        "reason": "Explicit Critical/Required/Nit labels on every finding, secret as Critical, PII as Required, `data` as Nit, ordered by leverage with a concrete fix per finding.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 47700,
          "output_tokens": 1621,
          "cached_tokens": 35055,
          "wall_ms": 28529
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.860833,
          "sd": 0.068388,
          "judge_sd_mean": 0.036603,
          "variance_ratio": 1.868371,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "d74271d29a7afb632beabd934a3c2f9f163bb6b2cb8e88581d3fa409a2e7c647",
              "status": "measured",
              "samples": [
                0.85,
                0.65,
                0.85
              ],
              "judge_sample_hashes": [
                "a786f1de283545307f0a87176fb1797202cd34f5d385d2209ac4ccf5070bc3a6",
                "4ecb14b20e8acd133cc84c16d22d53383ab3e9533d5da40f241d0efe1788dfb6",
                "47370f7922265ce6d1ceecce0e996050bf0f30401cea6b5c4578c80b8e652dd8"
              ],
              "mean": 0.783333,
              "stddev": 0.11547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 26813,
                "output_tokens": 2358,
                "cached_tokens": 26811,
                "wall_ms": 12546
              },
              "judge_usage": {
                "input_tokens": 47553,
                "output_tokens": 2235,
                "cached_tokens": 34957,
                "wall_ms": 38274
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "7b9b385590f05c93ba1faa94060d1556a0f07b05d247efa19d52879235b5aed6",
              "status": "measured",
              "samples": [
                0.82,
                0.8,
                0.85
              ],
              "judge_sample_hashes": [
                "4027f3535f2e351084b58a8a66956a80b22f9fef2fb2e0f8b1c009a0953a7e0c",
                "93a5730427065bce68e96e01d8c288cec1484df486aa67cae988ca4b724e9637",
                "2995352911d041b63d18548586dc8dc87de2aeaaa0b8089bba49ce15a6bcba21"
              ],
              "mean": 0.823333,
              "stddev": 0.025166,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 26813,
                "output_tokens": 2568,
                "cached_tokens": 26811,
                "wall_ms": 13258
              },
              "judge_usage": {
                "input_tokens": 47415,
                "output_tokens": 2898,
                "cached_tokens": 34865,
                "wall_ms": 43050
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "726d1046b7e5955a2345159a611d14f1338094c06c519786903a0d93e0f39e5e",
              "status": "measured",
              "samples": [
                0.91,
                0.92,
                0.92
              ],
              "judge_sample_hashes": [
                "0443b2861f54705e9dbfb79e9bd8d600c4e770dd36f1fd543d4a8408d505c2ff",
                "4f8a4386e550a13b17bdf3862faf4b9eded10ef168577c151efd572efa39e1bc",
                "5fbd1eb0361bf5fd9261b06c88ea72ba35218b3e1e4b6b781cfbec75e0390780"
              ],
              "mean": 0.916667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 26813,
                "output_tokens": 2717,
                "cached_tokens": 26811,
                "wall_ms": 14611
              },
              "judge_usage": {
                "input_tokens": 47568,
                "output_tokens": 1537,
                "cached_tokens": 34967,
                "wall_ms": 34141
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "250b0a366f5f45f355c5d778fd86d00ec8a3d49d271634da2b2b039189136ebb",
              "status": "measured",
              "samples": [
                0.92,
                0.92,
                0.92
              ],
              "judge_sample_hashes": [
                "e97123bce6e0dbdf18ad7c9afe3f4bbf7fee7e04f8fb483f1bec8bea9ff7f75b",
                "808872493befc82e61264e9a389414b0a33b6b30746d33872282f55bdb31c396",
                "47b624b95449595be4a76cee7917fbb537cfdb64872b83985abcca98c43998b2"
              ],
              "mean": 0.92,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 26813,
                "output_tokens": 2512,
                "cached_tokens": 26811,
                "wall_ms": 13363
              },
              "judge_usage": {
                "input_tokens": 47700,
                "output_tokens": 1621,
                "cached_tokens": 35055,
                "wall_ms": 28529
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.8125,
        "mean": 0.8125,
        "stddev": 0.076079,
        "samples": [
          0.86,
          0.85,
          0.86
        ],
        "generation_hash": "6865c47aa631ac7bb85e8f1b89c71bc879becd981272d3d01f36437d9c405b2b",
        "judge_sample_hashes": [
          "95afea0b4b3e4413d850ad66937a01f1499f172759db815c7e1cccf52438a942",
          "89442c80816f9c9b73d16cc628e774c054f3038558ffe6730d4dce5debeb5e54",
          "61694aa8192a006160f15f7aa15773a55db2da9e5fb2b321b321546e63b4e150"
        ],
        "threshold": 0.7,
        "reason": "Explicit severity tiers, secret correctly Critical, PII logging High, ordered by leverage with concrete fixes; uses High/Low instead of Required/Nit and omits the `data` naming nit.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 47232,
          "output_tokens": 2181,
          "cached_tokens": 34743,
          "wall_ms": 35934
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.8125,
          "sd": 0.076079,
          "judge_sd_mean": 0.036266,
          "variance_ratio": 2.097805,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "e6160f272a834c2f2785485f3271323054772e023cda772f753497e34ad1fe0a",
              "status": "measured",
              "samples": [
                0.88,
                0.9,
                0.89
              ],
              "judge_sample_hashes": [
                "d9c80fa3d3712c9fa8e91a788281deb6f5722ebc7e3261c0d7b6e1f7d6c17c3b",
                "8427451f7c6d4e4d365579befc976818f0316d2c735ad44fd0dc45895b82b2a8",
                "d41996e28c58fc48e3883a018fa0e79d3df5a5abc1894b18e7ad9d6755ebe07c"
              ],
              "mean": 0.89,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19838,
                "output_tokens": 2317,
                "cached_tokens": 19836,
                "wall_ms": 12139
              },
              "judge_usage": {
                "input_tokens": 47733,
                "output_tokens": 1591,
                "cached_tokens": 35077,
                "wall_ms": 27789
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "7249991e46fcaeaabaf51d61492112be02c9a1fb58ff785954a997d642cb315a",
              "status": "measured",
              "samples": [
                0.7,
                0.78,
                0.68
              ],
              "judge_sample_hashes": [
                "d63256c366fa818a73d04afba6a599c08f5a002a78e6950a7e9a1a3ed6994f29",
                "d70c203c98a98bb1892d72c61e159cc66d02c69eeab65b32c5dcde3c23eae7f0",
                "22c77c45004b1274febe629859b19cad942ab54e26e117d3a50a930126bdc8c7"
              ],
              "mean": 0.72,
              "stddev": 0.052915,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19838,
                "output_tokens": 2293,
                "cached_tokens": 19836,
                "wall_ms": 11838
              },
              "judge_usage": {
                "input_tokens": 47331,
                "output_tokens": 2068,
                "cached_tokens": 34809,
                "wall_ms": 43087
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "3cc0acb175ba454f7d3e33034eb558a63ec4b0b4fcf70ba20c631c4b18b81da9",
              "status": "measured",
              "samples": [
                0.85,
                0.8,
                0.7
              ],
              "judge_sample_hashes": [
                "a24505d3663e9930250337f9c2ee9cf12d20c38bbefa71db65f69839659441ea",
                "884f3e7890ab7d477b3c2601fb401b6ecd9070005c231d6c6486e4adcb7e1c2a",
                "b0f72c7aafe4839007454954b3162d2e00e91464930da09e890a61a8fc4a9985"
              ],
              "mean": 0.783333,
              "stddev": 0.076376,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19838,
                "output_tokens": 1711,
                "cached_tokens": 19836,
                "wall_ms": 9501
              },
              "judge_usage": {
                "input_tokens": 47175,
                "output_tokens": 2051,
                "cached_tokens": 34705,
                "wall_ms": 79977
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "6865c47aa631ac7bb85e8f1b89c71bc879becd981272d3d01f36437d9c405b2b",
              "status": "measured",
              "samples": [
                0.86,
                0.85,
                0.86
              ],
              "judge_sample_hashes": [
                "95afea0b4b3e4413d850ad66937a01f1499f172759db815c7e1cccf52438a942",
                "89442c80816f9c9b73d16cc628e774c054f3038558ffe6730d4dce5debeb5e54",
                "61694aa8192a006160f15f7aa15773a55db2da9e5fb2b321b321546e63b4e150"
              ],
              "mean": 0.856667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-5-5",
              "usage": {
                "input_tokens": 19838,
                "output_tokens": 2454,
                "cached_tokens": 19836,
                "wall_ms": 12877
              },
              "judge_usage": {
                "input_tokens": 47232,
                "output_tokens": 2181,
                "cached_tokens": 34743,
                "wall_ms": 35934
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.860833,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.8125,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.860833,
    "baseline_score": 0.8125,
    "delta": 0.048333,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "dc58b1feb593eefa771834a5c6a1b3dd1a63a3b9e1133a993eab396978afe05c",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 4,
      "mean_input_tokens": 26813,
      "mean_output_tokens": 2538.75,
      "mean_cost_usd_per_call": 0.003951,
      "median_wall_ms": 13310.5,
      "wall_ms_p25": 12902,
      "wall_ms_p75": 13987,
      "wall_ms_iqr": 1085
    },
    "baseline": {
      "call_count": 4,
      "mean_input_tokens": 19838,
      "mean_output_tokens": 2193.75,
      "mean_cost_usd_per_call": 0.003081,
      "median_wall_ms": 11988.5,
      "wall_ms_p25": 10669.5,
      "wall_ms_p75": 12508,
      "wall_ms_iqr": 1838.5
    },
    "skill_incremental_cost_usd_per_call": 0.00087,
    "skill_incremental_cost_usd_per_1k_calls": 0.87,
    "output_tokens_delta": 345,
    "median_wall_ms_delta": 1322,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.56971,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}