{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-haiku-4-5-20251001",
    "model_release_date": "2025-10-01",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T20:57:32.293Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-4-5-20251001",
      "reported_models": [
        "claude-haiku-4-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T20:57:32.293Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-4-5-20251001": {
          "input_per_mtok": 1,
          "output_per_mtok": 5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "borderline",
        "score": 0.6775,
        "mean": 0.6775,
        "stddev": 0.044085,
        "samples": [
          0.68,
          0.62,
          0.68
        ],
        "generation_hash": "2cdebe9f7215ba744e8da024a49f5a7105072c74d0013a0fa2254f708442d7c0",
        "judge_sample_hashes": [
          "cc882c0e8de87b267784109d46d222ae6f618a265e85d2e4a00d2f5b1699e6ea",
          "6a65086d7f92a983d1336707e63ceaabbe7aef9b36c231486d371c9ca8caa58e",
          "71faab5f8228e02d604bb287d009d6124f61387d12af1ef847d712e2d7e3ad67"
        ],
        "threshold": 0.7,
        "reason": "Explicit severity labels on all findings, hardcoded secret correctly Critical, PII logging Critical, good leverage ordering and concrete fixes; but omits the cosmetic `data` naming nit, never exercising the Nit/Optional tier.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 48243,
          "output_tokens": 1621,
          "cached_tokens": 35417,
          "wall_ms": 35590
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.6775,
          "sd": 0.044085,
          "judge_sd_mean": 0.034151,
          "variance_ratio": 1.290885,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "eec895d4ea3999ede24248ff4ec8862d7500dab6f83251b15debf3b9c9e85c55",
              "status": "measured",
              "samples": [
                0.75,
                0.78,
                0.7
              ],
              "judge_sample_hashes": [
                "c82346a85f43779a38f5bbce7b3d0e6bf4b93af900de17892e5d62e0cc8b82da",
                "81d880d7cdfaf980cf589175b8ff844e462442ad6f12c8e6ab1e370f510be905",
                "472ee800531d0ad3e9de2dd213ab03290f1a028db640d978a906d0eb9c9f1251"
              ],
              "mean": 0.743333,
              "stddev": 0.040415,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 18950,
                "output_tokens": 1203,
                "cached_tokens": 0,
                "wall_ms": 15145
              },
              "judge_usage": {
                "input_tokens": 46443,
                "output_tokens": 2104,
                "cached_tokens": 34217,
                "wall_ms": 49179
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "81f084549d41289b2d5071adc9b9355d04abf8d8cb6f2f9ec2229583fe55cf5e",
              "status": "measured",
              "samples": [
                0.7,
                0.65,
                0.6
              ],
              "judge_sample_hashes": [
                "2399842bee674ab0e641cef4f4829f51981d1a0ebd7ff1fd3d04f2a2ff3aa6d6",
                "3cdec24f527b4b3e59954ba8ae2352eb4b1dab109278de7c3743b21eae68aa1c",
                "5bbdbdf5243ff3a8f25de6aacd63af1b83865fac558ea3f445bbafa57deaf79b"
              ],
              "mean": 0.65,
              "stddev": 0.05,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 18950,
                "output_tokens": 1396,
                "cached_tokens": 18940,
                "wall_ms": 16216
              },
              "judge_usage": {
                "input_tokens": 46710,
                "output_tokens": 2094,
                "cached_tokens": 34395,
                "wall_ms": 35710
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "fc3442f3fe842240e23e8b7438edb92b7ae8c1671c892f3573336ad72f63a154",
              "status": "measured",
              "samples": [
                0.65,
                0.67,
                0.65
              ],
              "judge_sample_hashes": [
                "ed792f29040f0160353d11636b9b2c202c02e213af5f1ffdb297daa393b8e6cc",
                "406252c250088f7119670d48302dabbf4bcbb09cbfd17644a36eadd192423f33",
                "95880055b83cba63f4afcc4b4db0c119687b828654db4830468124016755b42f"
              ],
              "mean": 0.656667,
              "stddev": 0.011547,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 18950,
                "output_tokens": 1354,
                "cached_tokens": 18940,
                "wall_ms": 15055
              },
              "judge_usage": {
                "input_tokens": 46464,
                "output_tokens": 1281,
                "cached_tokens": 34231,
                "wall_ms": 73772
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "2cdebe9f7215ba744e8da024a49f5a7105072c74d0013a0fa2254f708442d7c0",
              "status": "measured",
              "samples": [
                0.68,
                0.62,
                0.68
              ],
              "judge_sample_hashes": [
                "cc882c0e8de87b267784109d46d222ae6f618a265e85d2e4a00d2f5b1699e6ea",
                "6a65086d7f92a983d1336707e63ceaabbe7aef9b36c231486d371c9ca8caa58e",
                "71faab5f8228e02d604bb287d009d6124f61387d12af1ef847d712e2d7e3ad67"
              ],
              "mean": 0.66,
              "stddev": 0.034641,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 18950,
                "output_tokens": 1766,
                "cached_tokens": 18940,
                "wall_ms": 19535
              },
              "judge_usage": {
                "input_tokens": 48243,
                "output_tokens": 1621,
                "cached_tokens": 35417,
                "wall_ms": 35590
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "fail",
        "score": 0.338889,
        "mean": 0.338889,
        "stddev": 0.034695,
        "samples": [
          0.35,
          0.35,
          0.35
        ],
        "generation_hash": "a26dd73725223e4893375c1bdd2bfb3d000e1fb669d13caeaaab9c143a510d0a",
        "judge_sample_hashes": [
          "cbf18aa466c1cdbc2684dd34092cb86acdbbc68b583694d533202d71e94a8886",
          "ac4a9d6a75991c34da65e919adb39f63dc38b39d010ddb042d777de5474414f8",
          "c3be82648108d074aec26a981f8d4101514c4753845a982782f2c2e6f5060604"
        ],
        "threshold": 0.7,
        "reason": "Only one section is a true severity label (Critical, correctly covering the hardcoded secret); Functional/Code Quality are categories not severities, no Nit/Optional tier exists, PII logging demoted to Code Quality, and the `data` naming nit is omitted.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 47493,
          "output_tokens": 2057,
          "cached_tokens": 34917,
          "wall_ms": 121257
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 3,
          "n_measured": 3,
          "n_unmeasured": 0,
          "stopping_reason": "min_reached",
          "mean": 0.338889,
          "sd": 0.034695,
          "judge_sd_mean": 0.019245,
          "variance_ratio": 1.802806,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "386ddf53736d6bc9e888bf752428cde28f43c09aa4023d2283b892ef6b8dd119",
              "status": "measured",
              "samples": [
                0.4,
                0.3,
                0.4
              ],
              "judge_sample_hashes": [
                "b95f0916cb56631c24b2ade03a3629c764cf4374b3e3e6dd54e3df62b194dcab",
                "bff3c248d02d1752295bf845ccf2a21d58c812df01e4b843ccc1810e2b52f69d",
                "ec6c34b87d9ad843bdb6e0868100fd5815f95e239a47daeeb46ab5b2cf8e3279"
              ],
              "mean": 0.366667,
              "stddev": 0.057735,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14097,
                "output_tokens": 1099,
                "cached_tokens": 0,
                "wall_ms": 13200
              },
              "judge_usage": {
                "input_tokens": 45798,
                "output_tokens": 2281,
                "cached_tokens": 33787,
                "wall_ms": 39111
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "fc1d5e0420a08cf0a7c0e811f9e6e53dc3210a710fac2d229ee219da7fb72f56",
              "status": "measured",
              "samples": [
                0.3,
                0.3,
                0.3
              ],
              "judge_sample_hashes": [
                "280d94fabc1f386bd7e77e837c929e2bec529aa14e46d5f5f313899cbdaed6ce",
                "c97232db39336eeffd6d7fa66f963fa7f18e76cc18576844c99e37cc3562bb9d",
                "86b4ab1f4c70fc13eb278790eb9f01e48f458fcd2b0090031279add111dd21ce"
              ],
              "mean": 0.3,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14097,
                "output_tokens": 1018,
                "cached_tokens": 14087,
                "wall_ms": 13031
              },
              "judge_usage": {
                "input_tokens": 46224,
                "output_tokens": 1700,
                "cached_tokens": 34071,
                "wall_ms": 34589
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "a26dd73725223e4893375c1bdd2bfb3d000e1fb669d13caeaaab9c143a510d0a",
              "status": "measured",
              "samples": [
                0.35,
                0.35,
                0.35
              ],
              "judge_sample_hashes": [
                "cbf18aa466c1cdbc2684dd34092cb86acdbbc68b583694d533202d71e94a8886",
                "ac4a9d6a75991c34da65e919adb39f63dc38b39d010ddb042d777de5474414f8",
                "c3be82648108d074aec26a981f8d4101514c4753845a982782f2c2e6f5060604"
              ],
              "mean": 0.35,
              "stddev": 0,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14097,
                "output_tokens": 1554,
                "cached_tokens": 14087,
                "wall_ms": 16904
              },
              "judge_usage": {
                "input_tokens": 47493,
                "output_tokens": 2057,
                "cached_tokens": 34917,
                "wall_ms": 121257
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.6775,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.338889,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.6775,
    "baseline_score": 0.338889,
    "delta": 0.338611,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "d1a00f6045586a6cb089a0b761a18b61d9f146b12ba4e8b5ec5b94d84555b1e7",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 4,
      "mean_input_tokens": 18950,
      "mean_output_tokens": 1429.75,
      "mean_cost_usd_per_call": 0.026099,
      "median_wall_ms": 15680.5,
      "wall_ms_p25": 15100,
      "wall_ms_p75": 17875.5,
      "wall_ms_iqr": 2775.5
    },
    "baseline": {
      "call_count": 3,
      "mean_input_tokens": 14097,
      "mean_output_tokens": 1223.67,
      "mean_cost_usd_per_call": 0.020215,
      "median_wall_ms": 13200,
      "wall_ms_p25": null,
      "wall_ms_p75": null,
      "wall_ms_iqr": null
    },
    "skill_incremental_cost_usd_per_call": 0.005884,
    "skill_incremental_cost_usd_per_1k_calls": 5.884,
    "output_tokens_delta": 206.08,
    "median_wall_ms_delta": 2480.5,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.57063,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}