{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-opus-5",
    "model_release_date": "2026-07-24",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-15T16:04:34.736Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5",
      "reported_models": [
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-15T16:04:34.736Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.917778,
        "mean": 0.917778,
        "stddev": 0.010183,
        "samples": [
          0.92,
          0.88,
          0.92
        ],
        "generation_hash": "a5dca229feefb09b3d3be5a9911a74e2bcc6e831320fdeb8e7c815cb74b0a052",
        "judge_sample_hashes": [
          "52bd3c3abae5cf8f29b10d9992eb01814e4354383c1d879eb742152cc2529815",
          "9c55b92289d054f8e3a5210253c852213436236501e882af6c0e68cf6b3de34d",
          "30d145a00a4c1033f2e0edce73f0c055b61ee8a83c9c74ace3c23f7810c3ebf0"
        ],
        "threshold": 0.7,
        "reason": "Critical/Required/Nit severity labels are explicit, and order runs Critical to nits with a concrete fix per finding. The hardcoded secret and PII are Critical, `data` is a Nit; section labels and 'FYI' are minor flaws.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 15588,
          "output_tokens": 1408,
          "cached_tokens": 12547,
          "wall_ms": 26685
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.917778,
          "sd": 0.010183,
          "judge_sd_mean": 0.009623,
          "variance_ratio": 1.058194,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "c0c8d227945e762f2c96a450ee5c392e97b0d8663520ef69406eea90326720f6",
              "status": "measured",
              "samples": [
                0.92,
                0.92,
                0.92
              ],
              "judge_sample_hashes": [
                "4ab52172097a6eceac900a6b189be30a9be19a2e9dd091f4e90204af7cb3f480",
                "18ab6af6166249b86b263fca1cd82205a6faaaa5f43e4e7afa340ad8a605074f",
                "c159e5322454612188588353d112629a631865c8c840eaf308f67418b0d55d0d"
              ],
              "mean": 0.92,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 9536,
                "output_tokens": 2182,
                "cached_tokens": 0,
                "wall_ms": 31428
              },
              "judge_usage": {
                "input_tokens": 15021,
                "output_tokens": 1062,
                "cached_tokens": 10010,
                "wall_ms": 23120
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "d2e08650b980ab6cb1e2ac9ed80ebfe384b803b6904cdf401bb773bc26559e50",
              "status": "measured",
              "samples": [
                0.92,
                0.93,
                0.93
              ],
              "judge_sample_hashes": [
                "ec7688e997cbd45f20473fd7c7e62f2177fd3930ba4f2dcb673c9dc9af7d6c41",
                "b8d15e3cc04c458be4b28966ee20e7454c3b1bb2e04d53bf7108f10b8f645a74",
                "818cb93643d46e707df572e562a329cd0c30e47a435051136da8eb72c1cd12ef"
              ],
              "mean": 0.926667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 9536,
                "output_tokens": 2987,
                "cached_tokens": 9534,
                "wall_ms": 38631
              },
              "judge_usage": {
                "input_tokens": 15900,
                "output_tokens": 832,
                "cached_tokens": 12755,
                "wall_ms": 19384
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "a5dca229feefb09b3d3be5a9911a74e2bcc6e831320fdeb8e7c815cb74b0a052",
              "status": "measured",
              "samples": [
                0.92,
                0.88,
                0.92
              ],
              "judge_sample_hashes": [
                "52bd3c3abae5cf8f29b10d9992eb01814e4354383c1d879eb742152cc2529815",
                "9c55b92289d054f8e3a5210253c852213436236501e882af6c0e68cf6b3de34d",
                "30d145a00a4c1033f2e0edce73f0c055b61ee8a83c9c74ace3c23f7810c3ebf0"
              ],
              "mean": 0.906667,
              "stddev": 0.023094,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 9536,
                "output_tokens": 2753,
                "cached_tokens": 9028,
                "wall_ms": 35391
              },
              "judge_usage": {
                "input_tokens": 15588,
                "output_tokens": 1408,
                "cached_tokens": 12547,
                "wall_ms": 26685
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "borderline",
        "score": 0.783333,
        "mean": 0.783333,
        "stddev": 0.09432,
        "samples": [
          0.85,
          0.86,
          0.85
        ],
        "generation_hash": "df20602f6de6e564d032ef24aa2da629f2cd2de621b037ff98bb3f67536c559a",
        "judge_sample_hashes": [
          "14dbcb8968e9b6f7b4acca634fe970e4b592979176f0c7b32233157f0fc3ee46",
          "b7f8d5e1b3e53aebe4d5dcacbe2856a859c5f549a0978e49eabeee493ab40063",
          "8a2442cc51962a130e0f42e2c55cb192659d371e91549d21466b283b46eefabd"
        ],
        "threshold": 0.7,
        "reason": "Severity tiers are explicit (Critical/High/Medium/Low) and ordered by leverage; the hardcoded secret is Critical, PII logging High, naming Low. The `data` name is lumped into Medium; labels deviate from skill vocabulary.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 14985,
          "output_tokens": 1362,
          "cached_tokens": 12145,
          "wall_ms": 25832
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.783333,
          "sd": 0.09432,
          "judge_sd_mean": 0.022604,
          "variance_ratio": 4.172713,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "463c44e4b90575667a00388aa2bb4bed0c52b83ab070b6c76c796047a59a49cd",
              "status": "measured",
              "samples": [
                0.8,
                0.75,
                0.8
              ],
              "judge_sample_hashes": [
                "b62c35b64bfe454a029f63d4920f3fa58fcb5a277bd6814af082f6d040d79b12",
                "11a563920332fdf665615c701c9abeb3e06d25ff88db3f3ec9e4092ed049b8ec",
                "937ed162a8ccbecacaf98ab05bb2b373a7e295dcaf559faccb332917c3787cc3"
              ],
              "mean": 0.783333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 2561,
                "output_tokens": 2138,
                "cached_tokens": 0,
                "wall_ms": 27696
              },
              "judge_usage": {
                "input_tokens": 15432,
                "output_tokens": 1286,
                "cached_tokens": 12443,
                "wall_ms": 24713
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "41c9f204614b803d00493f64da1a8953895b820326e8aa8b584e3a59fa3784d4",
              "status": "measured",
              "samples": [
                0.85,
                0.84,
                0.85
              ],
              "judge_sample_hashes": [
                "26bfcd50e54a63a55274eccefcd71d5c0d0e542ea4c7953119c1b26d3ebd547c",
                "71847586cee3f8fbea18e1d2328ab34ee65729b10319f91b15dee3816b51b65d",
                "7c4cd4e7d880702d233be974722293e4bdd481cd047c6f8eaa59f4c4b8065ff5"
              ],
              "mean": 0.846667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 2561,
                "output_tokens": 2103,
                "cached_tokens": 2559,
                "wall_ms": 27863
              },
              "judge_usage": {
                "input_tokens": 14988,
                "output_tokens": 1292,
                "cached_tokens": 12147,
                "wall_ms": 24544
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "8488f5dce988f04ee465a79ba7ad56a1845ac11d871a6daef757e62f70d34636",
              "status": "measured",
              "samples": [
                0.7,
                0.65,
                0.6
              ],
              "judge_sample_hashes": [
                "4149e8cb7b3ce049acc83bfbb0171674afd2f9badf44ebe674e74c5a1217375f",
                "1dc8e148196ec94d5d663cfcc7a9e80aec15d78d1b07ebd33c98762578e2a35e",
                "0a55e3a0944ef3da95d9e7e08243332534a0d407f7ebf2cbbce6c8cc27e4b524"
              ],
              "mean": 0.65,
              "stddev": 0.05,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 2561,
                "output_tokens": 1964,
                "cached_tokens": 2559,
                "wall_ms": 26139
              },
              "judge_usage": {
                "input_tokens": 14856,
                "output_tokens": 1428,
                "cached_tokens": 12059,
                "wall_ms": 27477
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "df20602f6de6e564d032ef24aa2da629f2cd2de621b037ff98bb3f67536c559a",
              "status": "measured",
              "samples": [
                0.85,
                0.86,
                0.85
              ],
              "judge_sample_hashes": [
                "14dbcb8968e9b6f7b4acca634fe970e4b592979176f0c7b32233157f0fc3ee46",
                "b7f8d5e1b3e53aebe4d5dcacbe2856a859c5f549a0978e49eabeee493ab40063",
                "8a2442cc51962a130e0f42e2c55cb192659d371e91549d21466b283b46eefabd"
              ],
              "mean": 0.853333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 2561,
                "output_tokens": 2055,
                "cached_tokens": 2559,
                "wall_ms": 26396
              },
              "judge_usage": {
                "input_tokens": 14985,
                "output_tokens": 1362,
                "cached_tokens": 12145,
                "wall_ms": 25832
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.917778,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.783333,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.917778,
    "baseline_score": 0.783333,
    "delta": 0.134445,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "4e3e1dddda5592ac7a54864995a19e33ea8d50ed72330a64cbcc9ee4c30b8f4f",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 9536,
      "mean_output_tokens": 2640.67,
      "mean_cost_usd_per_call": 0.113697,
      "median_wall_ms": 35391,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 4,
      "mean_input_tokens": 2561,
      "mean_output_tokens": 2065,
      "mean_cost_usd_per_call": 0.06443,
      "median_wall_ms": 27046,
      "wall_ms_p25": 26267.5,
      "wall_ms_p75": 27779.5,
      "wall_ms_iqr": 1512
    },
    "skill_incremental_cost_usd_per_call": 0.049267,
    "skill_incremental_cost_usd_per_1k_calls": 49.267,
    "output_tokens_delta": 575.67,
    "median_wall_ms_delta": 8345,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.222115,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}