{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-opus-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T06:14:31.245Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-opus-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-opus-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T06:14:31.245Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-opus-5-5": {
          "input_per_mtok": 4,
          "output_per_mtok": 20,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.903333,
        "mean": 0.903333,
        "stddev": 0.014529,
        "samples": [
          0.9,
          0.91,
          0.92
        ],
        "generation_hash": "5299673ca0a4cfde8f8c60699f21a12605ab3d2298e12b748a4d683b633fb369",
        "judge_sample_hashes": [
          "c738eccb9009218b013fcbfff25c1cb4261fcd5f799cae148357569dd78eded9",
          "b39081b772e6c1b6ca25f1e7b0753b32c20d1e168ca19196f334ada876eea7c0",
          "0b48790d4348b5b1c98b90261101ac8976fa8dd4450a584555826220b1f81861"
        ],
        "threshold": 0.7,
        "reason": "Explicit Critical/Required/Consider/Nit labels on every finding, secret marked Critical, PII logging Critical, `data` name a Nit, ordered by leverage with concrete fixes.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 48873,
          "output_tokens": 1443,
          "cached_tokens": 35833,
          "wall_ms": 31987
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.903333,
          "sd": 0.014529,
          "judge_sd_mean": 0.011031,
          "variance_ratio": 1.317106,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "bba0ff9e1c317f32d7bf00e8f9ade97b7124c6f2e76c3e0594a4b0e40d4ad7c3",
              "status": "measured",
              "samples": [
                0.9,
                0.88,
                0.88
              ],
              "judge_sample_hashes": [
                "a0c63413ef5396608ece6c9e48f00ccaa84fec9f71837edf6394cbc738173d64",
                "cbd1833609fb087558b356ca0c24c1b1ee598d4d6dc1e77e79df2b7c68590f5a",
                "5682cfa019268e56353d333596a4d8ad304b459593121e6b71354fb8049a4c42"
              ],
              "mean": 0.886667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 19712,
                "output_tokens": 1891,
                "cached_tokens": 19710,
                "wall_ms": 21060
              },
              "judge_usage": {
                "input_tokens": 48087,
                "output_tokens": 1900,
                "cached_tokens": 35309,
                "wall_ms": 31410
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "6e76c1a87e8799796124a8390fdd6139e26d7b5480019f88c2d2aed13bd68648",
              "status": "measured",
              "samples": [
                0.9,
                0.92,
                0.92
              ],
              "judge_sample_hashes": [
                "c1fbdec00057e524e87075e94a9e4964786085722fbb2c844d2b54c7be969416",
                "ba00139282ca3bf8c48fefef0896f1d9828eae47d7705b1f3a9f948777c92108",
                "9b408ba6a147fa736c3a99c145f025af2034d2247a3f933bd2f059b6ce94b0b6"
              ],
              "mean": 0.913333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 19712,
                "output_tokens": 1400,
                "cached_tokens": 19710,
                "wall_ms": 15423
              },
              "judge_usage": {
                "input_tokens": 47979,
                "output_tokens": 1963,
                "cached_tokens": 35237,
                "wall_ms": 34750
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "5299673ca0a4cfde8f8c60699f21a12605ab3d2298e12b748a4d683b633fb369",
              "status": "measured",
              "samples": [
                0.9,
                0.91,
                0.92
              ],
              "judge_sample_hashes": [
                "c738eccb9009218b013fcbfff25c1cb4261fcd5f799cae148357569dd78eded9",
                "b39081b772e6c1b6ca25f1e7b0753b32c20d1e168ca19196f334ada876eea7c0",
                "0b48790d4348b5b1c98b90261101ac8976fa8dd4450a584555826220b1f81861"
              ],
              "mean": 0.91,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 19712,
                "output_tokens": 2187,
                "cached_tokens": 19710,
                "wall_ms": 23863
              },
              "judge_usage": {
                "input_tokens": 48873,
                "output_tokens": 1443,
                "cached_tokens": 35833,
                "wall_ms": 31987
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "borderline",
        "score": 0.713333,
        "mean": 0.713333,
        "stddev": 0.211332,
        "samples": [
          0.88,
          0.88,
          0.86
        ],
        "generation_hash": "cd3c1543afbff809d5164cf97ae2fa6018cf787c9338a1193fd384bee09910cb",
        "judge_sample_hashes": [
          "23a286bb733c1c0caa00c6a08af5185300ecc76f1df087a81279855069a2e3d1",
          "874981f927d1aac6509e4b575b61b593a3b524de90b35c8cac8786588d28abce",
          "1540875a91b69f6099d131b8b874599f7a4e93fba66af9f9357f322355821f1f"
        ],
        "threshold": 0.7,
        "reason": "Every finding carries an explicit severity (Critical/High/Medium/Low), hardcoded secret and PII logging are Critical, cosmetic naming is Low, findings ordered by leverage, concrete fixes given; `data` naming folded into a Medium rather than labeled a nit.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 48468,
          "output_tokens": 2240,
          "cached_tokens": 35563,
          "wall_ms": 38675
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 5,
          "n_measured": 5,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.713333,
          "sd": 0.211332,
          "judge_sd_mean": 0.011238,
          "variance_ratio": 18.805125,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "4b8095e95959e1e179ef51cd5ed797124702e4a76e0bec3cf10509ce5d8f8315",
              "status": "measured",
              "samples": [
                0.85,
                0.86,
                0.87
              ],
              "judge_sample_hashes": [
                "7929cf444ebaa2634e41fb3b00bcd66a4836725f72b1746204cbfda427ba5499",
                "371f9a524b0bae44f83be1288237f43f2ed34e637625e45c9a11c4abf1e44524",
                "cf8a8c297cf944992f3ec79b361c43647ea9d1ed05a5894375868cf723d1dd61"
              ],
              "mean": 0.86,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12737,
                "output_tokens": 1408,
                "cached_tokens": 12735,
                "wall_ms": 14734
              },
              "judge_usage": {
                "input_tokens": 48036,
                "output_tokens": 2332,
                "cached_tokens": 35275,
                "wall_ms": 39669
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "2e860f6ff674ef24df97ffae471c6eee6215cb847f1fad9f714b965134a88aa6",
              "status": "measured",
              "samples": [
                0.87,
                0.87,
                0.86
              ],
              "judge_sample_hashes": [
                "a59d9faae6ca0d48b509b88166beb6372d14b770b1e46d706280db916bd02e4e",
                "bf39d26b0cd83f56d112aed97d22e268361565c0b1365fbbf5734bcc5d1669b1",
                "6c4a3b2f96423c7e5a1fd0a8d5b2834811427658f2f3be574cda9b61715d2685"
              ],
              "mean": 0.866667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12737,
                "output_tokens": 1582,
                "cached_tokens": 12735,
                "wall_ms": 16864
              },
              "judge_usage": {
                "input_tokens": 48603,
                "output_tokens": 2125,
                "cached_tokens": 35653,
                "wall_ms": 35086
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "f81d9f3ed8386cb835032042c4b02c9379ff4c8d23c931f38bafa422c54384dd",
              "status": "measured",
              "samples": [
                0.45,
                0.45,
                0.45
              ],
              "judge_sample_hashes": [
                "147c8e063f1ce933b261fac3863d58ac72d2250ae11b11544f6404ad27263bb2",
                "daaed31c23ba04ac81208ee3b6084e64848903c13da4348952efdcf119b966b1",
                "52716a63c6a1a7f5a5bf9275ab4214fae42d89339eb02e24a74812179c0a9721"
              ],
              "mean": 0.45,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12737,
                "output_tokens": 1592,
                "cached_tokens": 12735,
                "wall_ms": 16964
              },
              "judge_usage": {
                "input_tokens": 48633,
                "output_tokens": 2591,
                "cached_tokens": 35673,
                "wall_ms": 42362
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "deae506e5b739115960ba6c49b06ee76d484ad29800c5ea1e161ff2276a838b3",
              "status": "measured",
              "samples": [
                0.5,
                0.5,
                0.55
              ],
              "judge_sample_hashes": [
                "36bafa32977fde7d01083b9f5bfd75364d466e7ea952d15ab0064d83a3a66006",
                "e300898291528fe51b8b8b3ba31d93520e1b742b8752d61d64449fe8c1b07111",
                "52da11f1e263d92b4659a7aedac92c4dabb975a083e20cd2d3676f4e9ca9ffbc"
              ],
              "mean": 0.516667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12737,
                "output_tokens": 1439,
                "cached_tokens": 12735,
                "wall_ms": 15881
              },
              "judge_usage": {
                "input_tokens": 48174,
                "output_tokens": 2928,
                "cached_tokens": 35367,
                "wall_ms": 71062
              }
            },
            {
              "draw_index": 4,
              "generation_hash": "cd3c1543afbff809d5164cf97ae2fa6018cf787c9338a1193fd384bee09910cb",
              "status": "measured",
              "samples": [
                0.88,
                0.88,
                0.86
              ],
              "judge_sample_hashes": [
                "23a286bb733c1c0caa00c6a08af5185300ecc76f1df087a81279855069a2e3d1",
                "874981f927d1aac6509e4b575b61b593a3b524de90b35c8cac8786588d28abce",
                "1540875a91b69f6099d131b8b874599f7a4e93fba66af9f9357f322355821f1f"
              ],
              "mean": 0.873333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-opus-5-5",
              "usage": {
                "input_tokens": 12737,
                "output_tokens": 1551,
                "cached_tokens": 12735,
                "wall_ms": 17850
              },
              "judge_usage": {
                "input_tokens": 48468,
                "output_tokens": 2240,
                "cached_tokens": 35563,
                "wall_ms": 38675
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.903333,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.713333,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.903333,
    "baseline_score": 0.713333,
    "delta": 0.19,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "de6e30b2499278e8ec7ba67eaa4abdb7c0a6aee6baa3eebab3515eb34863bd96",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 19712,
      "mean_output_tokens": 1826,
      "mean_cost_usd_per_call": 0.115368,
      "median_wall_ms": 21060,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 5,
      "mean_input_tokens": 12737,
      "mean_output_tokens": 1514.4,
      "mean_cost_usd_per_call": 0.081236,
      "median_wall_ms": 16864,
      "wall_ms_p25": 15307.5,
      "wall_ms_p75": 17407,
      "wall_ms_iqr": 2099.5
    },
    "skill_incremental_cost_usd_per_call": 0.034132,
    "skill_incremental_cost_usd_per_1k_calls": 34.132,
    "output_tokens_delta": 311.6,
    "median_wall_ms_delta": 4196,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.57878,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}