{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-sonnet-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T04:42:56.786Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T04:42:56.786Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.9,
        "mean": 0.9,
        "stddev": 0.012018,
        "samples": [
          0.92,
          0.89,
          0.92
        ],
        "generation_hash": "69f1c85c6c70d11b6404bce6f1688d30fe1cdd8a607c17ae6e545436b087b561",
        "judge_sample_hashes": [
          "13609dcb0f83f421c93c16e0c908932c66638f9f2774248ac681ed0d8b366f6f",
          "6c996103f7967233af738d664d67cff6822788f2cd8c03b15c090c153ed7dcfb",
          "92dcbb85ffd0ed9822f2e272c59a95c2caeed4a4c647e6c72e45f70249f455e3"
        ],
        "threshold": 0.7,
        "reason": "Explicit severity labels throughout; hardcoded secret Critical, PII logging Required, `data` naming a Nit; ordered Critical-first with concrete fixes for every substantive finding.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 48441,
          "output_tokens": 1922,
          "cached_tokens": 35545,
          "wall_ms": 31864
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.9,
          "sd": 0.012018,
          "judge_sd_mean": 0.014714,
          "variance_ratio": 0.816773,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "ebc7ec41378a706b82ec860531e511406efafab8c6830c353141f7daba5cb87c",
              "status": "measured",
              "samples": [
                0.88,
                0.88,
                0.9
              ],
              "judge_sample_hashes": [
                "91635440ab944b695c13281f247871ade14ddfcb0474845d17d3728bbf16958e",
                "f989bf4ab4be64389798dbc81b97a79fe6bfe4d7f7cd3d8d35e75d9ec326627c",
                "03df6656e4d43e6d06ac33731e4b1ab56285d88e6cadfd5b49ea91fffb9904be"
              ],
              "mean": 0.886667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 19716,
                "output_tokens": 1487,
                "cached_tokens": 19714,
                "wall_ms": 12575
              },
              "judge_usage": {
                "input_tokens": 48318,
                "output_tokens": 2218,
                "cached_tokens": 35463,
                "wall_ms": 38595
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "afcec76062c858d18cd6193c7a5dca4fca816a6349be022364d48d7b1b20c10b",
              "status": "measured",
              "samples": [
                0.92,
                0.9,
                0.89
              ],
              "judge_sample_hashes": [
                "d753a8b97d876074b26122663d8c63c3dfb8e83654730b464d5d68e145d39c70",
                "0cd554b656937d242991db3aa93919fcd604ef32b391e90564167dde3f639a97",
                "7d1f0fe36d7bff2f81c7396fdac6757836e75f0672470183c8f5db2813b93abc"
              ],
              "mean": 0.903333,
              "stddev": 0.015275,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 19716,
                "output_tokens": 1616,
                "cached_tokens": 19714,
                "wall_ms": 14110
              },
              "judge_usage": {
                "input_tokens": 48705,
                "output_tokens": 1978,
                "cached_tokens": 35721,
                "wall_ms": 32605
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "69f1c85c6c70d11b6404bce6f1688d30fe1cdd8a607c17ae6e545436b087b561",
              "status": "measured",
              "samples": [
                0.92,
                0.89,
                0.92
              ],
              "judge_sample_hashes": [
                "13609dcb0f83f421c93c16e0c908932c66638f9f2774248ac681ed0d8b366f6f",
                "6c996103f7967233af738d664d67cff6822788f2cd8c03b15c090c153ed7dcfb",
                "92dcbb85ffd0ed9822f2e272c59a95c2caeed4a4c647e6c72e45f70249f455e3"
              ],
              "mean": 0.91,
              "stddev": 0.017321,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 19716,
                "output_tokens": 1528,
                "cached_tokens": 19714,
                "wall_ms": 13279
              },
              "judge_usage": {
                "input_tokens": 48441,
                "output_tokens": 1922,
                "cached_tokens": 35545,
                "wall_ms": 31864
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.848889,
        "mean": 0.848889,
        "stddev": 0.043504,
        "samples": [
          0.88,
          0.9,
          0.89
        ],
        "generation_hash": "ba6a808b10b56f1464ecd8c0b1aa86f59e433adac8462b2a2615d34e16e7dbcc",
        "judge_sample_hashes": [
          "04e0d2f6a47df7e40bab85b3f6176cee1a36e106d984cb44f17f471907954607",
          "8e1ffeb26206ec1ea8f570d61adf229c5d890116b00f30788ed82bef803cab86",
          "579293ed6927b15d25c3573c3421d869dcfb8159586ce1e4cc065f7a33cb0d07"
        ],
        "threshold": 0.7,
        "reason": "Explicit severity labels throughout, secret Critical, PII logging High, `data` naming Minor, ordered by leverage with concrete fixes; label vocabulary deviates slightly from skill's Required/Nit taxonomy.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 48087,
          "output_tokens": 1867,
          "cached_tokens": 35309,
          "wall_ms": 31797
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.848889,
          "sd": 0.043504,
          "judge_sd_mean": 0.041867,
          "variance_ratio": 1.0391,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "c7498a4194d3090474786207d45e41482584f225ddc50275c7ad9ef4f786d7e7",
              "status": "measured",
              "samples": [
                0.89,
                0.72,
                0.8
              ],
              "judge_sample_hashes": [
                "1caf8fb840a065e473d8d4091f671d410d7e2bb9b4e3eb29cc660a220aba6a8d",
                "bdc943eb5fc9c765e1a56c5b1b75e1e3e3e4ba0c54c244fc5b8d9301fc33b1b2",
                "48bad8ad59d2eaaa8a83d98c80054b709e88450256798e43ebbdafb5daf7eecc"
              ],
              "mean": 0.803333,
              "stddev": 0.085049,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12741,
                "output_tokens": 1347,
                "cached_tokens": 12739,
                "wall_ms": 11490
              },
              "judge_usage": {
                "input_tokens": 47898,
                "output_tokens": 2522,
                "cached_tokens": 35183,
                "wall_ms": 38841
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "9dcc79a78d7a90af6234dedeb3e7fe2047cbf672c32d2551237095ccce3b85b8",
              "status": "measured",
              "samples": [
                0.82,
                0.88,
                0.86
              ],
              "judge_sample_hashes": [
                "eb714f3e7834d106c1056bce4032d0a5f114a51f6b518ea664ac9270e5c1b80f",
                "43947b3ea7bddc938edda64e9b23aef015e7f04f332a18a3e41ea51172206fa1",
                "18e7ef49be7bb0f0512994b6c433c778dd797a622763492e3851418f2371c645"
              ],
              "mean": 0.853333,
              "stddev": 0.030551,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12741,
                "output_tokens": 1357,
                "cached_tokens": 12739,
                "wall_ms": 11908
              },
              "judge_usage": {
                "input_tokens": 47928,
                "output_tokens": 2210,
                "cached_tokens": 35203,
                "wall_ms": 42583
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "ba6a808b10b56f1464ecd8c0b1aa86f59e433adac8462b2a2615d34e16e7dbcc",
              "status": "measured",
              "samples": [
                0.88,
                0.9,
                0.89
              ],
              "judge_sample_hashes": [
                "04e0d2f6a47df7e40bab85b3f6176cee1a36e106d984cb44f17f471907954607",
                "8e1ffeb26206ec1ea8f570d61adf229c5d890116b00f30788ed82bef803cab86",
                "579293ed6927b15d25c3573c3421d869dcfb8159586ce1e4cc065f7a33cb0d07"
              ],
              "mean": 0.89,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12741,
                "output_tokens": 1410,
                "cached_tokens": 12739,
                "wall_ms": 12837
              },
              "judge_usage": {
                "input_tokens": 48087,
                "output_tokens": 1867,
                "cached_tokens": 35309,
                "wall_ms": 31797
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.9,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.848889,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.9,
    "baseline_score": 0.848889,
    "delta": 0.051111,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "67265be2b2c19393c7bdefb77918953ce441f0f1dfc3c58bcc58b6d81160baca",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 19716,
      "mean_output_tokens": 1543.67,
      "mean_cost_usd_per_call": 0.054869,
      "median_wall_ms": 13279,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 12741,
      "mean_output_tokens": 1371.33,
      "mean_cost_usd_per_call": 0.039195,
      "median_wall_ms": 11908,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.015674,
    "skill_incremental_cost_usd_per_1k_calls": 15.674,
    "output_tokens_delta": 172.34,
    "median_wall_ms_delta": 1371,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.577365,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}