{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-sonnet-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T05:42:54.981Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T05:42:54.981Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.895556,
        "mean": 0.895556,
        "stddev": 0.001925,
        "samples": [
          0.9,
          0.9,
          0.89
        ],
        "generation_hash": "e639426c3fa30d9a8e0ad978a66f0bb8c20e03f7e7ede06eb29406564e909250",
        "judge_sample_hashes": [
          "01a895609032a3c1bc8e1a0906f7818d8a2c79b761db7e1c202105ada75840a9",
          "54718ba9d2131bc98eed9ec2ec34a880b12af7388b3d615cc6fd75225deff90a",
          "79348c75136d8e49e4b88417801e71d5c6511e4ba19610a550b23f53ac0cd984"
        ],
        "threshold": 0.7,
        "reason": "Explicit Critical/Required/Optional labels, secret marked Critical, PII logging Required, `data` naming Optional, ordered by leverage with concrete fixes; severity tiers slightly broad (no Nit tier).",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 47970,
          "output_tokens": 1949,
          "cached_tokens": 35231,
          "wall_ms": 43205
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.895556,
          "sd": 0.001925,
          "judge_sd_mean": 0.007698,
          "variance_ratio": 0.250065,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "155876baff22510ac78b6abb4f07b6a73c52721d486c91956ad5d5ae55375968",
              "status": "measured",
              "samples": [
                0.88,
                0.9,
                0.9
              ],
              "judge_sample_hashes": [
                "0e720177e7812d8382f0e093e4ab371c4e6d358600fbf160ce4a4b2924ea9914",
                "893979c013d096d59c0a7483f192e74754f0a054fe3c4b513c8997022a8f9491",
                "fb5a95b218393c1f32234aa6d921353bfc37b30334ffcf613486b87bc36013c0"
              ],
              "mean": 0.893333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 19716,
                "output_tokens": 1385,
                "cached_tokens": 19714,
                "wall_ms": 13582
              },
              "judge_usage": {
                "input_tokens": 48012,
                "output_tokens": 1949,
                "cached_tokens": 35259,
                "wall_ms": 33576
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "8fb08af28bbb4223d32c24d584f71ce0c88ccfe61dae13f7ead828394ec5e1b2",
              "status": "measured",
              "samples": [
                0.9,
                0.89,
                0.9
              ],
              "judge_sample_hashes": [
                "8e834c68a3b971119802f23eb65f5dba357304955e50f2b1140624a0cdfc6aef",
                "e7b1416b666175b55268405825c16501fcd885754f30164f5624418b54ce8052",
                "358faa5389ff8a5d4f10bb75816a76f020bddb874d8b71947d27f68280f7806f"
              ],
              "mean": 0.896667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 19716,
                "output_tokens": 1437,
                "cached_tokens": 19714,
                "wall_ms": 13012
              },
              "judge_usage": {
                "input_tokens": 48168,
                "output_tokens": 1935,
                "cached_tokens": 35363,
                "wall_ms": 31574
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "e639426c3fa30d9a8e0ad978a66f0bb8c20e03f7e7ede06eb29406564e909250",
              "status": "measured",
              "samples": [
                0.9,
                0.9,
                0.89
              ],
              "judge_sample_hashes": [
                "01a895609032a3c1bc8e1a0906f7818d8a2c79b761db7e1c202105ada75840a9",
                "54718ba9d2131bc98eed9ec2ec34a880b12af7388b3d615cc6fd75225deff90a",
                "79348c75136d8e49e4b88417801e71d5c6511e4ba19610a550b23f53ac0cd984"
              ],
              "mean": 0.896667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 19716,
                "output_tokens": 1371,
                "cached_tokens": 19714,
                "wall_ms": 12458
              },
              "judge_usage": {
                "input_tokens": 47970,
                "output_tokens": 1949,
                "cached_tokens": 35231,
                "wall_ms": 43205
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.861111,
        "mean": 0.861111,
        "stddev": 0.025459,
        "samples": [
          0.87,
          0.88,
          0.85
        ],
        "generation_hash": "4c32b03dfd3f50474abae82ece0f9e0a93110dac0414e531a5be6b0ad15a6655",
        "judge_sample_hashes": [
          "ff5b71b1768985ce75260f4251c9e0ee13c186e201bff6511b2adb64926f4f38",
          "24bf7231d47f51a484c7ea3e080bf96bff247649b690357f62cb73fb10c0df29",
          "25a1ea42dfae4ec6110787d8763dfe82cc0d68e2724dc68ee4ca44a337de2dce"
        ],
        "threshold": 0.7,
        "reason": "Explicit severity tiers on every finding, hardcoded secret Critical, PII logging High, ordered by leverage with concrete fixes; cosmetic `data` naming rated Medium rather than Nit.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 48186,
          "output_tokens": 1930,
          "cached_tokens": 35375,
          "wall_ms": 35597
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.861111,
          "sd": 0.025459,
          "judge_sd_mean": 0.022769,
          "variance_ratio": 1.118143,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "e407d1560d7a3b95c9909c944ed67410731ead2d223dc2b9f25f41121e8ac1e1",
              "status": "measured",
              "samples": [
                0.85,
                0.78,
                0.87
              ],
              "judge_sample_hashes": [
                "25619b33f97f3fa85ccdebf80a00b4b13b8ee3796d767f635e073e326baac395",
                "894e4d525e5da0a4d41cf9fe786f434e8a2f645aff1c89555f722778b7d1ba89",
                "44ced682327b71577f598862a69d1ad96d33c31f8257e5e019af4cf6e5227100"
              ],
              "mean": 0.833333,
              "stddev": 0.047258,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12741,
                "output_tokens": 1316,
                "cached_tokens": 12739,
                "wall_ms": 14324
              },
              "judge_usage": {
                "input_tokens": 47805,
                "output_tokens": 2664,
                "cached_tokens": 35121,
                "wall_ms": 41971
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "1a794ebb0be0a6fa8f82d7815ea9194b1569f1fa54447cc6c3eb873ec7dd64a4",
              "status": "measured",
              "samples": [
                0.89,
                0.88,
                0.88
              ],
              "judge_sample_hashes": [
                "4000732261a818e3dc87a4db2854692eb91f052f2b92954561319bf23c3446f2",
                "3dbdfcaf39df269cdea353cd6994e5190593b42424c948366fb6a022a617b23e",
                "7d7dd2aa0f24bbad339fba0b385ad694a1528b9208c457c7ebf45605bade56f3"
              ],
              "mean": 0.883333,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12741,
                "output_tokens": 1197,
                "cached_tokens": 12739,
                "wall_ms": 10698
              },
              "judge_usage": {
                "input_tokens": 47448,
                "output_tokens": 1900,
                "cached_tokens": 34883,
                "wall_ms": 30229
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "4c32b03dfd3f50474abae82ece0f9e0a93110dac0414e531a5be6b0ad15a6655",
              "status": "measured",
              "samples": [
                0.87,
                0.88,
                0.85
              ],
              "judge_sample_hashes": [
                "ff5b71b1768985ce75260f4251c9e0ee13c186e201bff6511b2adb64926f4f38",
                "24bf7231d47f51a484c7ea3e080bf96bff247649b690357f62cb73fb10c0df29",
                "25a1ea42dfae4ec6110787d8763dfe82cc0d68e2724dc68ee4ca44a337de2dce"
              ],
              "mean": 0.866667,
              "stddev": 0.015275,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12741,
                "output_tokens": 1443,
                "cached_tokens": 12739,
                "wall_ms": 14040
              },
              "judge_usage": {
                "input_tokens": 48186,
                "output_tokens": 1930,
                "cached_tokens": 35375,
                "wall_ms": 35597
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.895556,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.861111,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.895556,
    "baseline_score": 0.861111,
    "delta": 0.034445,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "d7e4c01f343915d7fd91bcc236d81a6bac7d39f3ea9806f6c852d99b46703be6",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 19716,
      "mean_output_tokens": 1397.67,
      "mean_cost_usd_per_call": 0.053409,
      "median_wall_ms": 13012,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 12741,
      "mean_output_tokens": 1318.67,
      "mean_cost_usd_per_call": 0.038669,
      "median_wall_ms": 14040,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.01474,
    "skill_incremental_cost_usd_per_1k_calls": 14.74,
    "output_tokens_delta": 79,
    "median_wall_ms_delta": -1028,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.577755,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}