{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-opus-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-23T06:34:35.169Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-opus-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-23T06:34:35.169Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5-5": {
          "input_per_mtok": 4,
          "output_per_mtok": 20,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.91,
        "mean": 0.91,
        "stddev": 0.01453,
        "samples": [
          0.91,
          0.92,
          0.92
        ],
        "generation_hash": "0ff0fc9c7de3d8338ae4b763e2ed00842f23b26a3d40fa25eaa1940dc2f4d8b3",
        "judge_sample_hashes": [
          "9718509373333869d75bf4cb0d54718977ce500d61ef4f5f598178672a58864f",
          "0bd34147d8471ea90ab7f0eda9fdcb95edf05aef3cd4049d55dfe7e198372894",
          "1a9e5ac9956e421b0fdeba4b881ee44e9b7a08f862da99844a4f34bdd854da10"
        ],
        "threshold": 0.7,
        "reason": "Explicit Critical/Required/Nit labels on every finding; secret is Critical, PII logging Required, `data` name a Nit; ordered by leverage with concrete fixes throughout.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 21408,
          "output_tokens": 1980,
          "cached_tokens": 17524,
          "wall_ms": 33578
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.91,
          "sd": 0.01453,
          "judge_sd_mean": 0.005774,
          "variance_ratio": 2.516453,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "0c2f1f9697871f5c57c80cd0da1a0a2cc01933a73e3a88043abfaaabcc1e1362",
              "status": "measured",
              "samples": [
                0.92,
                0.92,
                0.92
              ],
              "judge_sample_hashes": [
                "9af33a12260faf82c0cd3ef2886ec87bb035ff353744a23ac1ecfd5999f1f4df",
                "1f150121f9e08e7d11b10af1f2a97d33dd9d52759efed0e884795fc88d6f5f2a",
                "5b69242c591f848b17ddbe836c22686a86d4316a998bb82d4c7e0c9d0620d088"
              ],
              "mean": 0.92,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 10646,
                "output_tokens": 1840,
                "cached_tokens": 540,
                "wall_ms": 19427
              },
              "judge_usage": {
                "input_tokens": 21207,
                "output_tokens": 1892,
                "cached_tokens": 14674,
                "wall_ms": 29098
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "ace5cbe44526af2926f0ed53262f145d9d66d3aff932d20c6fbd2fd37567972a",
              "status": "measured",
              "samples": [
                0.9,
                0.9,
                0.88
              ],
              "judge_sample_hashes": [
                "442dd3d9ae8ba4f0b057ed1f1811d56e81106a6da450f4754f8c538380ede518",
                "0778c70bdd14b83bc42f19c30125d64b83c380e97c45442190ec795578513baa",
                "31238b463244417d880cdb32f01d667df5f273abcf4b542bf91229db45efabc8"
              ],
              "mean": 0.893333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 10646,
                "output_tokens": 1973,
                "cached_tokens": 10644,
                "wall_ms": 22085
              },
              "judge_usage": {
                "input_tokens": 21252,
                "output_tokens": 2214,
                "cached_tokens": 17420,
                "wall_ms": 33401
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "0ff0fc9c7de3d8338ae4b763e2ed00842f23b26a3d40fa25eaa1940dc2f4d8b3",
              "status": "measured",
              "samples": [
                0.91,
                0.92,
                0.92
              ],
              "judge_sample_hashes": [
                "9718509373333869d75bf4cb0d54718977ce500d61ef4f5f598178672a58864f",
                "0bd34147d8471ea90ab7f0eda9fdcb95edf05aef3cd4049d55dfe7e198372894",
                "1a9e5ac9956e421b0fdeba4b881ee44e9b7a08f862da99844a4f34bdd854da10"
              ],
              "mean": 0.916667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 10646,
                "output_tokens": 1923,
                "cached_tokens": 10644,
                "wall_ms": 20987
              },
              "judge_usage": {
                "input_tokens": 21408,
                "output_tokens": 1980,
                "cached_tokens": 17524,
                "wall_ms": 33578
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "borderline",
        "score": 0.68,
        "mean": 0.68,
        "stddev": 0.217682,
        "samples": [
          0.87,
          0.88,
          0.88
        ],
        "generation_hash": "78fca214ca6ad3ebd05877e06626c068f61d30de9d4a8ff3e903a1904c54141e",
        "judge_sample_hashes": [
          "3d704522edba7d349aa996c0b2b6a384c1715663f956f27c89af9403fdb1864f",
          "0f99f462ae9ce92b8e2dab9ed1800626bc70bf969ae01c1b9f32ff4b911cce25",
          "7751adb4a465d757bc386951ad3dc46be01d586bb236678668e871b6ac1b5910"
        ],
        "threshold": 0.7,
        "reason": "Explicit severity sections with hardcoded secret as Critical, PII logging High, naming as Medium/Low; ordered by leverage with concrete fixes, but skips the `data` name nit explicitly.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 20760,
          "output_tokens": 2489,
          "cached_tokens": 17092,
          "wall_ms": 40482
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.68,
          "sd": 0.217682,
          "judge_sd_mean": 0.030877,
          "variance_ratio": 7.049972,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "910f1e1d76832538465e4ba2d466cc9981cc7afe2aeb9eb9815259d8acdd9ea8",
              "status": "measured",
              "samples": [
                0.87,
                0.85,
                0.86
              ],
              "judge_sample_hashes": [
                "372372fc3c785e106b398f584228de9805c39970ed77c50a8e9abc9a5b5abd3d",
                "7e9b418cc6c9b2f8a118eb710e4241cee537994de6ee4e5c19933f7122da9718",
                "46b54a087c3d8caef8377cf2f06eed0a34c55945dc8ecb3d3ce8f3df088cced0"
              ],
              "mean": 0.86,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 3671,
                "output_tokens": 1456,
                "cached_tokens": 531,
                "wall_ms": 15488
              },
              "judge_usage": {
                "input_tokens": 21027,
                "output_tokens": 2006,
                "cached_tokens": 17270,
                "wall_ms": 33654
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "b84788384889d96273dfb3187246a9492f5dd6d1545ad6ecf896828985ec4935",
              "status": "measured",
              "samples": [
                0.55,
                0.45,
                0.45
              ],
              "judge_sample_hashes": [
                "0551ec8e26a05d4bdfc16c854ff6bb8fcfe515bdbfa1d2adfcad7416d10da015",
                "052b584a720f1aa88029721e9a011077b4fc5877e3b9283e1c97834e777a910d",
                "8cfa0a7d51f27c9929fae5b381c165b6ccfe4aed319789b7097c362696d3535c"
              ],
              "mean": 0.483333,
              "stddev": 0.057735,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 3671,
                "output_tokens": 1538,
                "cached_tokens": 3669,
                "wall_ms": 16301
              },
              "judge_usage": {
                "input_tokens": 21273,
                "output_tokens": 3371,
                "cached_tokens": 17434,
                "wall_ms": 49263
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "5d45527413e0c3a689fefa0d9f037c8bbecfd2ba7565937771d7d48586e10d95",
              "status": "measured",
              "samples": [
                0.45,
                0.5,
                0.55
              ],
              "judge_sample_hashes": [
                "fb575c8b2d84ffac29e639070e0406b29cd22f9ec7253b6cc051d876f42ad446",
                "0e3208cb3745436fbbc0e51223fa173eb5628abfe99b2e4c3223ddaa920276d6",
                "57c5266e8898d6e69e2ca96a08760cc83ae2bd72a5b22e58623edd7bed0c370c"
              ],
              "mean": 0.5,
              "stddev": 0.05,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 3671,
                "output_tokens": 1445,
                "cached_tokens": 3669,
                "wall_ms": 14432
              },
              "judge_usage": {
                "input_tokens": 20994,
                "output_tokens": 3144,
                "cached_tokens": 17248,
                "wall_ms": 45724
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "78fca214ca6ad3ebd05877e06626c068f61d30de9d4a8ff3e903a1904c54141e",
              "status": "measured",
              "samples": [
                0.87,
                0.88,
                0.88
              ],
              "judge_sample_hashes": [
                "3d704522edba7d349aa996c0b2b6a384c1715663f956f27c89af9403fdb1864f",
                "0f99f462ae9ce92b8e2dab9ed1800626bc70bf969ae01c1b9f32ff4b911cce25",
                "7751adb4a465d757bc386951ad3dc46be01d586bb236678668e871b6ac1b5910"
              ],
              "mean": 0.876667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 3671,
                "output_tokens": 1367,
                "cached_tokens": 3669,
                "wall_ms": 14682
              },
              "judge_usage": {
                "input_tokens": 20760,
                "output_tokens": 2489,
                "cached_tokens": 17092,
                "wall_ms": 40482
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.91,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.68,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.91,
    "baseline_score": 0.68,
    "delta": 0.23,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "5c3ad19b40dd4b39bd73899709528f3547a77e84abf43e0707d3b6a4d76ea3b3",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 10646,
      "mean_output_tokens": 1912,
      "mean_cost_usd_per_call": 0.080824,
      "median_wall_ms": 20987,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 4,
      "mean_input_tokens": 3671,
      "mean_output_tokens": 1451.5,
      "mean_cost_usd_per_call": 0.043714,
      "median_wall_ms": 15085,
      "wall_ms_p25": 14557,
      "wall_ms_p75": 15894.5,
      "wall_ms_iqr": 1337.5
    },
    "skill_incremental_cost_usd_per_call": 0.03711,
    "skill_incremental_cost_usd_per_1k_calls": 37.11,
    "output_tokens_delta": 460.5,
    "median_wall_ms_delta": 5902,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.322565,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}