{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-sonnet-5-5",
    "model_release_date": null,
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-09-29T03:57:59.530Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-sonnet-5-5",
      "reported_models": [
        "claude-opus-5",
        "claude-sonnet-5-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-09-29T03:57:59.530Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-sonnet-5-5": {
          "input_per_mtok": 2,
          "output_per_mtok": 10,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "pass",
        "score": 0.886666,
        "mean": 0.886666,
        "stddev": 0.025166,
        "samples": [
          0.88,
          0.87,
          0.9
        ],
        "generation_hash": "2144f0ca8b969e2238b8ca3f99815ac8f999ce16004883cb84f19222ed0a94b5",
        "judge_sample_hashes": [
          "1159a3a81a98f0fd28a29e5bb6a98947ac8ca57ac8b212da359b61b67d9ded2d",
          "801558aff9a78e3696dd55a3bdd43f69743c47d21a7d45c4b0658c6592785d62",
          "82c54af27212587982befef7d13e5ce5f89ee59f2313af2a2435498ebfce05b8"
        ],
        "threshold": 0.7,
        "reason": "Explicit Critical/Required/Consider labels on every finding, secret marked Critical, PII logging Required, `data` name in Consider, ordered by leverage with concrete fixes; un-awaited fetch slightly over-labeled Critical.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 47835,
          "output_tokens": 1966,
          "cached_tokens": 35141,
          "wall_ms": 33641
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.886666,
          "sd": 0.025166,
          "judge_sd_mean": 0.01279,
          "variance_ratio": 1.967631,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "0fd7d9149662ddffd3c67ebd38b25f35548b5a8f2a7c8cf08b48df1296f861bd",
              "status": "measured",
              "samples": [
                0.85,
                0.87,
                0.87
              ],
              "judge_sample_hashes": [
                "75e59ec3732af1f365fcf0f9c3f5b679b410049648fc13acf2b33ceeb8c27a74",
                "f30d42b24043ead9a3ffdaf366562c8e3b6a6d8bf9947ac49310494269516ac7",
                "50bf8fcae1e55b59636a543b04f9f2617764ce6fbd03991c702b4e03eae188aa"
              ],
              "mean": 0.863333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 19716,
                "output_tokens": 1377,
                "cached_tokens": 540,
                "wall_ms": 13229
              },
              "judge_usage": {
                "input_tokens": 47988,
                "output_tokens": 2810,
                "cached_tokens": 32528,
                "wall_ms": 44238
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "0b2c407dfc6b187227b374baef047a7693ca574517e4c7ee5636ca9ffb1bae74",
              "status": "measured",
              "samples": [
                0.9,
                0.92,
                0.92
              ],
              "judge_sample_hashes": [
                "a6866066b4d6ce43a941ff30f75ee4a88ad68927981a084dc7b5bfd20c892277",
                "d162adc3be412d1fc4582fb9a59d3e2c0d2917a385ae98c6a910f253457b5b7f",
                "b67cc1f0fe8841ef015ba23b8e83d26db43385ac2b54de837cd3aee77d04d7f6"
              ],
              "mean": 0.913333,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 19716,
                "output_tokens": 1434,
                "cached_tokens": 19714,
                "wall_ms": 13418
              },
              "judge_usage": {
                "input_tokens": 48159,
                "output_tokens": 1658,
                "cached_tokens": 35357,
                "wall_ms": 25775
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "2144f0ca8b969e2238b8ca3f99815ac8f999ce16004883cb84f19222ed0a94b5",
              "status": "measured",
              "samples": [
                0.88,
                0.87,
                0.9
              ],
              "judge_sample_hashes": [
                "1159a3a81a98f0fd28a29e5bb6a98947ac8ca57ac8b212da359b61b67d9ded2d",
                "801558aff9a78e3696dd55a3bdd43f69743c47d21a7d45c4b0658c6592785d62",
                "82c54af27212587982befef7d13e5ce5f89ee59f2313af2a2435498ebfce05b8"
              ],
              "mean": 0.883333,
              "stddev": 0.015275,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 19716,
                "output_tokens": 1326,
                "cached_tokens": 19714,
                "wall_ms": 13088
              },
              "judge_usage": {
                "input_tokens": 47835,
                "output_tokens": 1966,
                "cached_tokens": 35141,
                "wall_ms": 33641
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "pass",
        "score": 0.877778,
        "mean": 0.877778,
        "stddev": 0.027148,
        "samples": [
          0.89,
          0.9,
          0.9
        ],
        "generation_hash": "0cac28f9025f76f7d20c01793ac837b8959449fdbef77c00d9e13fefb407d036",
        "judge_sample_hashes": [
          "3f613dfc23de0d44a66117b2cfc3a7508ce856ec804dd0a3b81a65dc519a0889",
          "171b9ac0e48560246277ff4b0a7e983a44f5211574a2ef2be7b2f6407625feec",
          "4eb57af8a6dfd2873bc0ecc2063eb06f4a9f37f1518fa901ddf6122b96f29e30"
        ],
        "threshold": 0.7,
        "reason": "Every finding carries an explicit severity heading, secret is Critical, PII logging High, `data` name Low; ordered by leverage with concrete fixes, but labels deviate from skill's Required/Nit vocabulary.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 48102,
          "output_tokens": 1848,
          "cached_tokens": 35319,
          "wall_ms": 30277
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.877778,
          "sd": 0.027148,
          "judge_sd_mean": 0.01873,
          "variance_ratio": 1.449439,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "6a382562f184c32387f9376c113d0972553ba06d89e6c295db782b16933e1046",
              "status": "measured",
              "samples": [
                0.9,
                0.89,
                0.88
              ],
              "judge_sample_hashes": [
                "66ad81ebdd2efdc40c90ed70554f141d96b909d5df98a477d5f8962823f83f00",
                "0b95249ee83e742e00d758a0966d8ef180c0a48e10620d7baa60a980f8fb9e5b",
                "5eccb851d7d0e9914810afadaaec2a9a857f20689efccf34a3f5a59143904d5e"
              ],
              "mean": 0.89,
              "stddev": 0.01,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12741,
                "output_tokens": 1415,
                "cached_tokens": 531,
                "wall_ms": 12632
              },
              "judge_usage": {
                "input_tokens": 48102,
                "output_tokens": 1520,
                "cached_tokens": 35319,
                "wall_ms": 26788
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "f9f9416da690c2ebcd5df176df3a08354bb171788d3a878f3b172d2e324e42fc",
              "status": "measured",
              "samples": [
                0.87,
                0.8,
                0.87
              ],
              "judge_sample_hashes": [
                "809eac80a3c9b07903d7c03461cedf4e550d0bf16a455d9a8f77f5576f8f9288",
                "6d1b7e1029e84a72dc4e0f325ac5894e7db8f089fa7844b057954dc7793b58ba",
                "d9c7750e070e1f697d79e23ffa88f4f3b368fecebf8040b9ab5623b8b958e04b"
              ],
              "mean": 0.846667,
              "stddev": 0.040415,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12741,
                "output_tokens": 1246,
                "cached_tokens": 12739,
                "wall_ms": 10645
              },
              "judge_usage": {
                "input_tokens": 47595,
                "output_tokens": 2497,
                "cached_tokens": 34981,
                "wall_ms": 39621
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "0cac28f9025f76f7d20c01793ac837b8959449fdbef77c00d9e13fefb407d036",
              "status": "measured",
              "samples": [
                0.89,
                0.9,
                0.9
              ],
              "judge_sample_hashes": [
                "3f613dfc23de0d44a66117b2cfc3a7508ce856ec804dd0a3b81a65dc519a0889",
                "171b9ac0e48560246277ff4b0a7e983a44f5211574a2ef2be7b2f6407625feec",
                "4eb57af8a6dfd2873bc0ecc2063eb06f4a9f37f1518fa901ddf6122b96f29e30"
              ],
              "mean": 0.896667,
              "stddev": 0.005774,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-sonnet-5-5",
              "usage": {
                "input_tokens": 12741,
                "output_tokens": 1415,
                "cached_tokens": 12739,
                "wall_ms": 12184
              },
              "judge_usage": {
                "input_tokens": 48102,
                "output_tokens": 1848,
                "cached_tokens": 35319,
                "wall_ms": 30277
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.886666,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 1,
        "borderline_count": 0,
        "mean_score": 0.877778,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.886666,
    "baseline_score": 0.877778,
    "delta": 0.008888,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "1910eae70ac29aad695ff1a36b68c4d9318126637ce93e9697331539b14ac3f4",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 3,
      "mean_input_tokens": 19716,
      "mean_output_tokens": 1379,
      "mean_cost_usd_per_call": 0.053222,
      "median_wall_ms": 13229,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 12741,
      "mean_output_tokens": 1358.67,
      "mean_cost_usd_per_call": 0.039069,
      "median_wall_ms": 12184,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.014153,
    "skill_incremental_cost_usd_per_1k_calls": 14.153,
    "output_tokens_delta": 20.33,
    "median_wall_ms_delta": 1045,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.575035,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}