{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-sonnet-5",
    "model_release_date": "2026-06-30",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T04:56:13.861Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T04:56:13.861Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.881111,
        "mean": 0.881111,
        "stddev": 0.005092,
        "samples": [
          0.9,
          0.88,
          0.88
        ],
        "generation_hash": "49eaf9b320ab29fae2b37e3055470632b7c528cf438021e9c0ab3acdf67538f4",
        "judge_sample_hashes": [
          "720e672f1164d279ce43bdc9a36eb4b2da028cb12ec0a937355771a2a004ea04",
          "4dcbe2ef0a1801a4579430bc78c2689b10c53f1e9b45ebb5b7aa347eaaeb91af",
          "4988fc3c75e2494a4eed53ee54cc7331693597a4190dd232dcaf7e79f2a5b5d7"
        ],
        "threshold": 0.7,
        "reason": "Explicit Critical/Required/Nit labels; secret marked Critical, PII logging Critical, `data` name Nit; ordered by leverage with concrete rewrite, though slight duplication/over-labeling of some Required items.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 47253,
          "output_tokens": 1850,
          "cached_tokens": 34753,
          "wall_ms": 34484
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.881111,
          "sd": 0.005092,
          "judge_sd_mean": 0.009107,
          "variance_ratio": 0.55913,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "522b66274cd47f5465d32719271850c7448632af8ffea39d66660e970b5a97b3",
              "status": "measured",
              "samples": [
                0.88,
                0.88,
                0.87
              ],
              "judge_sample_hashes": [
                "c6f167cbb3b13c12864facd9d63cf8c05e7daf0136d7b08978cd286447901146",
                "fdba242847479e831de7b5040a6096ca41e1b7223025f73abe0aba06b2ceda38",
                "ae45689327f285109af01b6b528f72cf284d70e09d02ddc51e45a2facd407bfb"
              ],
              "mean": 0.876667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 26134,
                "output_tokens": 919,
                "cached_tokens": 26132,
                "wall_ms": 14590
              },
              "judge_usage": {
                "input_tokens": 46614,
                "output_tokens": 1742,
                "cached_tokens": 34327,
                "wall_ms": 41522
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "0df6b2153fb55ea3a721d068f578626a58eb876883c0ed28709854d682aad79d",
              "status": "measured",
              "samples": [
                0.89,
                0.88,
                0.87
              ],
              "judge_sample_hashes": [
                "bfcf06cd9435e97623e0ae66dcd1860576214ea7741da8080650bb2b508804f4",
                "c90e77fa8583b691c66c624ba5b6946bde1f3e58437f20fd3e294440cb8fa7ed",
                "5f1b502d78ab5f5d9595ed0386e1b678870e43f661e7aa2be1c8a08a36b18725"
              ],
              "mean": 0.88,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 26134,
                "output_tokens": 970,
                "cached_tokens": 26132,
                "wall_ms": 13686
              },
              "judge_usage": {
                "input_tokens": 46767,
                "output_tokens": 1127,
                "cached_tokens": 34429,
                "wall_ms": 27637
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "49eaf9b320ab29fae2b37e3055470632b7c528cf438021e9c0ab3acdf67538f4",
              "status": "measured",
              "samples": [
                0.9,
                0.88,
                0.88
              ],
              "judge_sample_hashes": [
                "720e672f1164d279ce43bdc9a36eb4b2da028cb12ec0a937355771a2a004ea04",
                "4dcbe2ef0a1801a4579430bc78c2689b10c53f1e9b45ebb5b7aa347eaaeb91af",
                "4988fc3c75e2494a4eed53ee54cc7331693597a4190dd232dcaf7e79f2a5b5d7"
              ],
              "mean": 0.886667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 26134,
                "output_tokens": 1132,
                "cached_tokens": 26132,
                "wall_ms": 14203
              },
              "judge_usage": {
                "input_tokens": 47253,
                "output_tokens": 1850,
                "cached_tokens": 34753,
                "wall_ms": 34484
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "borderline",
        "score": 0.674444,
        "mean": 0.674444,
        "stddev": 0.036868,
        "samples": [
          0.72,
          0.7,
          0.72
        ],
        "generation_hash": "0407012fb9d186713686f2b8a7812960cf4554d8a7c686d41948bd7a0aa17bf8",
        "judge_sample_hashes": [
          "d495042691ce45a1af8bcff85510166cf701774886815e3b11ffe0eea4c92c42",
          "bf58b8e2f5bc677662e5fb5570646758080b74fa075219a07c1d69b29543e24f",
          "230af94fe6c73806edcd6a2aa88fee487044169b09dcafbceb1d80407c305f44"
        ],
        "threshold": 0.7,
        "reason": "Explicit severity labels on every finding, hardcoded secret marked Critical, PII logging High, ordered by leverage; but omits the cosmetic `data` naming Nit and never uses a lowest tier.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 46995,
          "output_tokens": 2144,
          "cached_tokens": 34581,
          "wall_ms": 41199
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.674444,
          "sd": 0.036868,
          "judge_sd_mean": 0.015396,
          "variance_ratio": 2.394648,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "5a7cdf7aa33aac6d2ebd65b247ab095c2b8beecd2d84aee1e227ee3e1c804230",
              "status": "measured",
              "samples": [
                0.68,
                0.68,
                0.65
              ],
              "judge_sample_hashes": [
                "f3eb329a2ccf6c691d2ebf00b90cad792f63423dd5575944c496d32f94fc533b",
                "35188ad5a5234d64e1f4ad1ae3c8e4f58efbbc83435cc4f1e7ff1c98d9e0d553",
                "23e5de98fbc47c562f1fe2a9d5c7ad9ce7bd78c7868937027ee678e08ad9ffcd"
              ],
              "mean": 0.67,
              "stddev": 0.017321,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19159,
                "output_tokens": 914,
                "cached_tokens": 19157,
                "wall_ms": 10755
              },
              "judge_usage": {
                "input_tokens": 46599,
                "output_tokens": 1781,
                "cached_tokens": 34317,
                "wall_ms": 33747
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "a006e3cf67b5fb972cee3ab9f6ee48f282df9e5daff53ca1863afdf72ed7f0ef",
              "status": "measured",
              "samples": [
                0.65,
                0.65,
                0.62
              ],
              "judge_sample_hashes": [
                "69cc84996a1c2191742ef70d0424747b9fefb478a9b4c278fc43e4039b4e47df",
                "0aedeba1898fc34858f11dea9c408e7088cdaa05c0848f499811641549425c26",
                "4a83d614f1e693366a10b35224bc70062e3cdb64f2ad84d8ad24ad40f06d1ec6"
              ],
              "mean": 0.64,
              "stddev": 0.017321,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19159,
                "output_tokens": 1091,
                "cached_tokens": 19157,
                "wall_ms": 14115
              },
              "judge_usage": {
                "input_tokens": 47130,
                "output_tokens": 1890,
                "cached_tokens": 34671,
                "wall_ms": 33409
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "0407012fb9d186713686f2b8a7812960cf4554d8a7c686d41948bd7a0aa17bf8",
              "status": "measured",
              "samples": [
                0.72,
                0.7,
                0.72
              ],
              "judge_sample_hashes": [
                "d495042691ce45a1af8bcff85510166cf701774886815e3b11ffe0eea4c92c42",
                "bf58b8e2f5bc677662e5fb5570646758080b74fa075219a07c1d69b29543e24f",
                "230af94fe6c73806edcd6a2aa88fee487044169b09dcafbceb1d80407c305f44"
              ],
              "mean": 0.713333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19159,
                "output_tokens": 1046,
                "cached_tokens": 19157,
                "wall_ms": 13098
              },
              "judge_usage": {
                "input_tokens": 46995,
                "output_tokens": 2144,
                "cached_tokens": 34581,
                "wall_ms": 41199
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.881111,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.674444,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.881111,
    "baseline_score": 0.674444,
    "delta": 0.206667,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "d46994612b0a2d5b729eed6448ed17b88fa72670b1b0f543efa1bfe60cbd9867",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 26134,
      "mean_output_tokens": 1007,
      "mean_cost_usd_per_call": 0.062338,
      "median_wall_ms": 14203,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 19159,
      "mean_output_tokens": 1017,
      "mean_cost_usd_per_call": 0.048488,
      "median_wall_ms": 13098,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.01385,
    "skill_incremental_cost_usd_per_1k_calls": 13.85,
    "output_tokens_delta": -10,
    "median_wall_ms_delta": 1105,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.57109,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}