{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-sonnet-5",
    "model_release_date": "2026-06-30",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T04:12:15.305Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T04:12:15.305Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.88,
        "mean": 0.88,
        "stddev": 0.005774,
        "samples": [
          0.88,
          0.88,
          0.9
        ],
        "generation_hash": "1335b0b73a4d8297e831b854ec08045cff71eea23ecf4a94f7da8531ef04e541",
        "judge_sample_hashes": [
          "777693ac1a43c53adeaffe488ee449fb731d10f59ad0f63f7d1ad3344b5c1910",
          "5e70736684bbd9d7d6013d9715e183589745fef09523ccccd4a701086af41f0e",
          "da530e8dd49f38e7c673b894b5819c1bcdda3a893fa9a7b55778f6ed51bad643"
        ],
        "threshold": 0.7,
        "reason": "Explicit Critical/Required/Nit labels on every finding, hardcoded secret Critical, PII logging Critical, `data` name Nit, ordered by leverage with concrete fixes; not quite flawless-exceptional.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 46737,
          "output_tokens": 1433,
          "cached_tokens": 34409,
          "wall_ms": 27370
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.88,
          "sd": 0.005774,
          "judge_sd_mean": 0.007698,
          "variance_ratio": 0.750065,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "9282457e4aa2e87b5d24f6424b9ca176ddea93c7f8974ddb2ff9ad411f474d19",
              "status": "measured",
              "samples": [
                0.88,
                0.87,
                0.88
              ],
              "judge_sample_hashes": [
                "c6dd89e8445226fae66c6ecac9149aefb3f06facb7770e4f2fe005c3a4ec6238",
                "a0839ad2bd22e55f3f1e9a1ce9783e478334546e997f70a251ee88029b5e0852",
                "d735f893ef3a628cdb8a1efd1ea0e7bb18b0555fe0efd5a760f26013e3f2e202"
              ],
              "mean": 0.876667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 26134,
                "output_tokens": 730,
                "cached_tokens": 3406,
                "wall_ms": 10573
              },
              "judge_usage": {
                "input_tokens": 46047,
                "output_tokens": 2175,
                "cached_tokens": 33949,
                "wall_ms": 37634
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "f8ae1b8bd9cfd84de32af4bd14b8adabbc5aae7261de5e97a3ff31d0fc86a147",
              "status": "measured",
              "samples": [
                0.88,
                0.88,
                0.87
              ],
              "judge_sample_hashes": [
                "453aad57576dab4dce3f3e191419774c75a3d914750be263a8cfe0e2fb0f050a",
                "9b237dc339c27e9625023c900007a3c2527ef0b60a616868d60402d910592cf5",
                "a4a7f9a802ade92ff71ede7fed070791e66186e11bb252746aca6ab1aae37c65"
              ],
              "mean": 0.876667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 26134,
                "output_tokens": 730,
                "cached_tokens": 26132,
                "wall_ms": 11919
              },
              "judge_usage": {
                "input_tokens": 46047,
                "output_tokens": 1129,
                "cached_tokens": 33949,
                "wall_ms": 31883
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "1335b0b73a4d8297e831b854ec08045cff71eea23ecf4a94f7da8531ef04e541",
              "status": "measured",
              "samples": [
                0.88,
                0.88,
                0.9
              ],
              "judge_sample_hashes": [
                "777693ac1a43c53adeaffe488ee449fb731d10f59ad0f63f7d1ad3344b5c1910",
                "5e70736684bbd9d7d6013d9715e183589745fef09523ccccd4a701086af41f0e",
                "da530e8dd49f38e7c673b894b5819c1bcdda3a893fa9a7b55778f6ed51bad643"
              ],
              "mean": 0.886667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 26134,
                "output_tokens": 960,
                "cached_tokens": 26132,
                "wall_ms": 16039
              },
              "judge_usage": {
                "input_tokens": 46737,
                "output_tokens": 1433,
                "cached_tokens": 34409,
                "wall_ms": 27370
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "borderline",
        "score": 0.578334,
        "mean": 0.578334,
        "stddev": 0.145615,
        "samples": [
          0.68,
          0.73,
          0.68
        ],
        "generation_hash": "12cff3c87d3281dc5227a77188e1ccbc4c412dff5d77061f4f2df961fe9b91f7",
        "judge_sample_hashes": [
          "f779366bbbd819d3936336023a194234cf4ca5e27c524539713865a7f1efa1bb",
          "34515c65418f071085c8962d2bb8376a3b14fac0b0c246567456c09c87ee5ac2",
          "1d3e98c4005c826b35accdf9446084a9d3741cd80e0a14ac8acd4aa887140dcc"
        ],
        "threshold": 0.7,
        "reason": "Explicit severity labels on every finding with secret and PII logging as Critical and leverage-ordered, but no low-severity finding at all — the `data` naming nit is absent.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 46551,
          "output_tokens": 1877,
          "cached_tokens": 34285,
          "wall_ms": 32242
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.578334,
          "sd": 0.145615,
          "judge_sd_mean": 0.051365,
          "variance_ratio": 2.834907,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "f5d37cd3ec419b5d5971c098ccf8a8f3898490c9587175304735fea380decf42",
              "status": "measured",
              "samples": [
                0.68,
                0.6,
                0.55
              ],
              "judge_sample_hashes": [
                "ca5cc1f6efc080213091bf09d5d58e0143cf24ffb408efc39348ba92c3dccbef",
                "6a84e8daa6b19ced83d84a64c21953cc5cc487aeb6710a5f932434100e3de5ff",
                "95002d182f667e319be5a567fb4678f7e778046a319885ffdc7f03ff790b9747"
              ],
              "mean": 0.61,
              "stddev": 0.065574,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19159,
                "output_tokens": 915,
                "cached_tokens": 3397,
                "wall_ms": 10509
              },
              "judge_usage": {
                "input_tokens": 46602,
                "output_tokens": 2707,
                "cached_tokens": 34319,
                "wall_ms": 49533
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "679fb04c9652d4e01b46f6c7b62220a4ec390799aaede8605cf11a34e6d3cb7d",
              "status": "measured",
              "samples": [
                0.68,
                0.62,
                0.62
              ],
              "judge_sample_hashes": [
                "870cea69616559bbe156d4ca5f7a5f423f70321a85ff12e61603abdf48c8ab38",
                "a49314da54e419171dd9db4a7f6c2b53dc97536c4a06ac2de7e52edc652583e8",
                "7884656ad0fbc32e136795028022f7c1c96c7d6c86c15f4336a32b8674cd1a11"
              ],
              "mean": 0.64,
              "stddev": 0.034641,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19159,
                "output_tokens": 1167,
                "cached_tokens": 19157,
                "wall_ms": 16750
              },
              "judge_usage": {
                "input_tokens": 47358,
                "output_tokens": 2350,
                "cached_tokens": 34823,
                "wall_ms": 39909
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "ce29e436b337dfaff0f59261fa7bcba3457ab7e1949f241a29fef42207704b1d",
              "status": "measured",
              "samples": [
                0.45,
                0.35,
                0.3
              ],
              "judge_sample_hashes": [
                "2a2470f02b88f5a7aad7a4ae37211c34c7595210373d46b575ea0bb1d3150493",
                "e828f112ad35f082ffcad6308fcf9acfeb5648f0be3db39916b81e29a82ac9d7",
                "fa7c1abeccc1ed37af3aacdc5a0d3c7da0f768a04e7873011c59f470ddb6622f"
              ],
              "mean": 0.366667,
              "stddev": 0.076376,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19159,
                "output_tokens": 1182,
                "cached_tokens": 19157,
                "wall_ms": 15052
              },
              "judge_usage": {
                "input_tokens": 47403,
                "output_tokens": 2157,
                "cached_tokens": 34853,
                "wall_ms": 35332
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "12cff3c87d3281dc5227a77188e1ccbc4c412dff5d77061f4f2df961fe9b91f7",
              "status": "measured",
              "samples": [
                0.68,
                0.73,
                0.68
              ],
              "judge_sample_hashes": [
                "f779366bbbd819d3936336023a194234cf4ca5e27c524539713865a7f1efa1bb",
                "34515c65418f071085c8962d2bb8376a3b14fac0b0c246567456c09c87ee5ac2",
                "1d3e98c4005c826b35accdf9446084a9d3741cd80e0a14ac8acd4aa887140dcc"
              ],
              "mean": 0.696667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5",
              "usage": {
                "input_tokens": 19159,
                "output_tokens": 898,
                "cached_tokens": 19157,
                "wall_ms": 11705
              },
              "judge_usage": {
                "input_tokens": 46551,
                "output_tokens": 1877,
                "cached_tokens": 34285,
                "wall_ms": 32242
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.88,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.578334,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.88,
    "baseline_score": 0.578334,
    "delta": 0.301666,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "fbf686932b16a0b5ca909ddb8453c0da1c43850a772958e62aacea6d0fdc318c",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 26134,
      "mean_output_tokens": 806.67,
      "mean_cost_usd_per_call": 0.060335,
      "median_wall_ms": 11919,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 4,
      "mean_input_tokens": 19159,
      "mean_output_tokens": 1040.5,
      "mean_cost_usd_per_call": 0.048723,
      "median_wall_ms": 13378.5,
      "wall_ms_p25": 11107,
      "wall_ms_p75": 15901,
      "wall_ms_iqr": 4794
    },
    "skill_incremental_cost_usd_per_call": 0.011612,
    "skill_incremental_cost_usd_per_1k_calls": 11.612,
    "output_tokens_delta": -233.83,
    "median_wall_ms_delta": -1459.5,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.54919,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}