{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-opus-5",
    "model_release_date": "2026-07-24",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-23T06:49:59.857Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5",
      "reported_models": [
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-23T06:49:59.857Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.901111,
        "mean": 0.901111,
        "stddev": 0.006939,
        "samples": [
          0.9,
          0.91,
          0.9
        ],
        "generation_hash": "f30ce2191e572daa2c80b957885f91aa58dd3b72a2b2aa640e9c16fc63410dbe",
        "judge_sample_hashes": [
          "3dd1b2c80dbfd0d0814a020e0dd44fe64a199df833191ba4b533bfbd373532d2",
          "cbc7e011731565855a5045aa966754139576df4dd876cca56f1c857368f77544",
          "279ff24728a005bf72336be927ad3cbb96f68c030263eace60cf246fcc9e1d24"
        ],
        "threshold": 0.7,
        "reason": "Explicit Critical/Required/Nit labels on every finding, secret marked Critical, PII logging Critical, `data` a Nit, ordered by leverage with concrete fixes; one Required item filed under Nit/Optional.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 23520,
          "output_tokens": 1817,
          "cached_tokens": 18932,
          "wall_ms": 28414
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.901111,
          "sd": 0.006939,
          "judge_sd_mean": 0.013472,
          "variance_ratio": 0.515068,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "312957bb519a42da1560552198ebd7b80e8f9e8bc21eb33fa53598a10d7ed110",
              "status": "measured",
              "samples": [
                0.9,
                0.9,
                0.88
              ],
              "judge_sample_hashes": [
                "8e9f4e12068255d47a84c7f8e1dea2c1622a8709cd2dedfb88fbb24679cd96a7",
                "9bc66546db95c253c5b0213bdd3524a35c06a17c68273809b06301115d1b8d87",
                "9f8065fba1832adc7f9e60b99c8aae28bce7d989560fed6d217ed18f6332d9fc"
              ],
              "mean": 0.893333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 11646,
                "output_tokens": 2642,
                "cached_tokens": 540,
                "wall_ms": 35035
              },
              "judge_usage": {
                "input_tokens": 21231,
                "output_tokens": 1942,
                "cached_tokens": 17406,
                "wall_ms": 30133
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "51493a9502dcbfdd3730085c5f3df5d95495b41a2fee1766598a2b27bc73c293",
              "status": "measured",
              "samples": [
                0.88,
                0.92,
                0.92
              ],
              "judge_sample_hashes": [
                "5a01038f6995ccc6f7ccb3d4485abe39c661350ac7da104c22fe22c499ae389d",
                "138677a9267a62121a7b328ea699b3ac85d270dd07a642bd81d253c50b559bf4",
                "8dd87bdf8b9e9eee161eb582e998714a642e2eee17623b688eaa254ad23bbb2b"
              ],
              "mean": 0.906667,
              "stddev": 0.023094,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 11646,
                "output_tokens": 2883,
                "cached_tokens": 11644,
                "wall_ms": 36932
              },
              "judge_usage": {
                "input_tokens": 21582,
                "output_tokens": 1620,
                "cached_tokens": 17640,
                "wall_ms": 27341
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "f30ce2191e572daa2c80b957885f91aa58dd3b72a2b2aa640e9c16fc63410dbe",
              "status": "measured",
              "samples": [
                0.9,
                0.91,
                0.9
              ],
              "judge_sample_hashes": [
                "3dd1b2c80dbfd0d0814a020e0dd44fe64a199df833191ba4b533bfbd373532d2",
                "cbc7e011731565855a5045aa966754139576df4dd876cca56f1c857368f77544",
                "279ff24728a005bf72336be927ad3cbb96f68c030263eace60cf246fcc9e1d24"
              ],
              "mean": 0.903333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 11646,
                "output_tokens": 3751,
                "cached_tokens": 11644,
                "wall_ms": 48223
              },
              "judge_usage": {
                "input_tokens": 23520,
                "output_tokens": 1817,
                "cached_tokens": 18932,
                "wall_ms": 28414
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.862222,
        "mean": 0.862222,
        "stddev": 0.025019,
        "samples": [
          0.88,
          0.88,
          0.87
        ],
        "generation_hash": "c1356e324411fc2bcd291606dff9acdaa98b910322aa1ba81927d89b4e97bb77",
        "judge_sample_hashes": [
          "7174c51dbbfaa07a1ccfd023dcc1d2c0a9d68624677d81753ab58a25e08a647b",
          "9d9b35690b42ae59e3cf0596b08fae85ecb805af39c6bdaa7e22254db03b1d5a",
          "10103e0639bdc8de6dcfdf5bdd39e5e87d4c498d43fec4e98f2cbd9aff6589cf"
        ],
        "threshold": 0.7,
        "reason": "Every finding carries an explicit severity, secret is Critical, PII logging High, `data` naming Low, ordered by leverage with concrete fixes; labels deviate from the skill's Required/Nit vocabulary.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 22494,
          "output_tokens": 1939,
          "cached_tokens": 18248,
          "wall_ms": 29075
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.862222,
          "sd": 0.025019,
          "judge_sd_mean": 0.013472,
          "variance_ratio": 1.857111,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "0f35c9ef409f9fa7501f4d91b86a393bafcc1e0f41a3ce93be2997e264a1fb46",
              "status": "measured",
              "samples": [
                0.88,
                0.88,
                0.87
              ],
              "judge_sample_hashes": [
                "abce0374bab7e6dd913dcf1f03425735bf213a74a8d3427508d808ffc20ea876",
                "389096e20b3fe27c4f36294ad0a80e783a5a5c653ef3eaf23d138febfecca92e",
                "c35fef2748097a4c10f8caa805cd05363a18b319078aa7e22643a4e2350a4154"
              ],
              "mean": 0.876667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 4671,
                "output_tokens": 2682,
                "cached_tokens": 531,
                "wall_ms": 36266
              },
              "judge_usage": {
                "input_tokens": 22740,
                "output_tokens": 1799,
                "cached_tokens": 18412,
                "wall_ms": 28555
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "2feea343cec4c585b388de81319089eb10c94279209716e436365430d1ddf2c2",
              "status": "measured",
              "samples": [
                0.8,
                0.85,
                0.85
              ],
              "judge_sample_hashes": [
                "b5a121eb862ee520e4dd2cdc7bd0b77f7709733f577aff4f9cb2f11691e7806d",
                "14ae67c46df572c1741018e21879891f3bd5c2b5404822c0867f86d02521d550",
                "2a2306f32cff9ca7353bce34966233db78e8ed8ecfcbac7a5b855faf04f5b53b"
              ],
              "mean": 0.833333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 4671,
                "output_tokens": 2513,
                "cached_tokens": 4669,
                "wall_ms": 32637
              },
              "judge_usage": {
                "input_tokens": 21927,
                "output_tokens": 2494,
                "cached_tokens": 17870,
                "wall_ms": 37500
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "c1356e324411fc2bcd291606dff9acdaa98b910322aa1ba81927d89b4e97bb77",
              "status": "measured",
              "samples": [
                0.88,
                0.88,
                0.87
              ],
              "judge_sample_hashes": [
                "7174c51dbbfaa07a1ccfd023dcc1d2c0a9d68624677d81753ab58a25e08a647b",
                "9d9b35690b42ae59e3cf0596b08fae85ecb805af39c6bdaa7e22254db03b1d5a",
                "10103e0639bdc8de6dcfdf5bdd39e5e87d4c498d43fec4e98f2cbd9aff6589cf"
              ],
              "mean": 0.876667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5",
              "usage": {
                "input_tokens": 4671,
                "output_tokens": 2560,
                "cached_tokens": 4669,
                "wall_ms": 37518
              },
              "judge_usage": {
                "input_tokens": 22494,
                "output_tokens": 1939,
                "cached_tokens": 18248,
                "wall_ms": 29075
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.901111,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.862222,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.901111,
    "baseline_score": 0.862222,
    "delta": 0.038889,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "d7bd2fcd3dbea881a53581022cea4a311ed9e75b457aa6d91a7c4d1c3aaaecd0",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 11646,
      "mean_output_tokens": 3092,
      "mean_cost_usd_per_call": 0.13553,
      "median_wall_ms": 36932,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 4671,
      "mean_output_tokens": 2585,
      "mean_cost_usd_per_call": 0.08798,
      "median_wall_ms": 36266,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.04755,
    "skill_incremental_cost_usd_per_1k_calls": 47.55,
    "output_tokens_delta": 507,
    "median_wall_ms_delta": 666,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.32397,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}