{
  "schema_version": "0.6",
  "generation_sampled": true,
  "skill": {
    "name": "code-review-and-quality",
    "version": "0.0.0",
    "content_hash": "13d360d7f786de371886e404b1eaa0cd37ecabab90e5c7126d62d42b73a872ea",
    "tokens": 5120
  },
  "suite": {
    "format": "agentskills.io/evals",
    "suite_hash": "5d729f8852949735a553218a63ade5e0b5634b17579117290866bf074b9c6b40",
    "case_count": 1,
    "canary": "4ceca5ae-96c9-7547-9798-11d14894349f"
  },
  "run": {
    "model_id": "claude-haiku-4-5-20251001",
    "model_release_date": "2025-10-01",
    "provider": "anthropic",
    "surface": "claude-cli",
    "runner_version": "0.10.1",
    "date_utc": "2026-10-07T21:45:31.864Z",
    "registry": "registered",
    "transcripts": "retained-local",
    "judge": {
      "samples": 3,
      "temperature": null,
      "sampling": "surface-controlled",
      "surface": "claude-cli",
      "model_id": "claude-opus-5",
      "prompt_template_hash": "82586d1e44f84df2b8ce5f18ac777866465cb5ad9686503c93d9776c93c344ed"
    },
    "answered_by": {
      "kind": "model",
      "attested": true,
      "reported_model": "claude-haiku-4-5-20251001",
      "reported_models": [
        "claude-haiku-4-5",
        "claude-opus-5"
      ],
      "isolation": "same-user"
    },
    "pricing_snapshot": {
      "frozen_at": "2026-10-07T21:45:31.864Z",
      "source": "config/models.json",
      "currency": "USD",
      "models": {
        "claude-haiku-4-5-20251001": {
          "input_per_mtok": 1,
          "output_per_mtok": 5,
          "registered": true
        },
        "claude-opus-5": {
          "input_per_mtok": 5,
          "output_per_mtok": 25,
          "registered": true
        }
      },
      "note": "Prices frozen at run time. Every derived cost in this receipt is computed from THIS snapshot, never from the live registry, so the receipt keeps its meaning when registry prices later change."
    }
  },
  "results": {
    "cases": [
      {
        "id": "severity-labeled-findings",
        "mode": "with_skill",
        "outcome": "fail",
        "score": 0.565834,
        "mean": 0.565834,
        "stddev": 0.110567,
        "samples": [
          0.62,
          0.67,
          0.65
        ],
        "generation_hash": "845cbc44b24aae36f61267aca86622b5880c2574ef0eaf0d901f0d13c8e59080",
        "judge_sample_hashes": [
          "5db569ec9a4acf8227866ccee8dc517284fbd5bb8cb7d2079d0f4399d20b3fc0",
          "4a7c790fc275c05ac670319b96768121fec6b5d362b5253fd123eabe39630aa0",
          "0c5f35a2a52cb361d0ae143a39eb18a251a7e3f5402a1958135d9ac00ccd957f"
        ],
        "threshold": 0.7,
        "reason": "Explicit severity labels throughout, secret correctly Critical, `data` name Optional, findings ordered by leverage; but PII console.log miscalibrated as Optional rather than Required/Critical (-0.2).",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 47106,
          "output_tokens": 1574,
          "cached_tokens": 34659,
          "wall_ms": 32653
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.565834,
          "sd": 0.110567,
          "judge_sd_mean": 0.027943,
          "variance_ratio": 3.956876,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "19bd7c4e4500a0c0d05abd272f4a832de57fb18d8830e1c9edad2702b723ff51",
              "status": "measured",
              "samples": [
                0.65,
                0.7,
                0.65
              ],
              "judge_sample_hashes": [
                "505baca8eee743775ac7b4ddc02142dd35e1f7fec72c14fc98430d54bfcbcee1",
                "90a5c56e022437b3b9aaa2d361db4b63a7e405f0b75e25b17bdbc82ce39b0e7f",
                "a5f7837dd7f6038a11ca2711f9a6864af4e4851b1f36ec515321a9881855abb1"
              ],
              "mean": 0.666667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 18950,
                "output_tokens": 902,
                "cached_tokens": 18940,
                "wall_ms": 13080
              },
              "judge_usage": {
                "input_tokens": 45390,
                "output_tokens": 1755,
                "cached_tokens": 33515,
                "wall_ms": 32231
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "46c0f01f3237c7530cb07a5818d73f416af7579bbbe9c5ddb8450b39fc2cc1a7",
              "status": "measured",
              "samples": [
                0.45,
                0.45,
                0.4
              ],
              "judge_sample_hashes": [
                "c73a484ae408a553784335b00f4804801f4dd5eee8dbc7292cab1e469fd1b4e9",
                "55259d9849bffed8ba3d90e01c87e2ff1ed9a6d04e83ef4cec8d2c1e1b1da480",
                "1c46e19a0bfcf85e48b2a872556654253dc3fe4bcd4852e6f98dd962f82e776d"
              ],
              "mean": 0.433333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 18950,
                "output_tokens": 1076,
                "cached_tokens": 18940,
                "wall_ms": 13836
              },
              "judge_usage": {
                "input_tokens": 46767,
                "output_tokens": 2701,
                "cached_tokens": 34433,
                "wall_ms": 88493
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "5ef858721d24739028fe38b7b846386124624f0c75e5566b69c78742e5378da3",
              "status": "measured",
              "samples": [
                0.5,
                0.55,
                0.5
              ],
              "judge_sample_hashes": [
                "4d7e5721b3d95bd21a46174a256d8a0ae0fc57195f7875f2acf698a1da0f4dc0",
                "cf520835535113bf20763e31cde7957523e0ecee27a30e09878f79a6ba68cc65",
                "e245b4a015c1afceab2c2fefa230b24605934f55d3fa118db21810f7a33fda3d"
              ],
              "mean": 0.516667,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 18950,
                "output_tokens": 1385,
                "cached_tokens": 18940,
                "wall_ms": 16329
              },
              "judge_usage": {
                "input_tokens": 47205,
                "output_tokens": 2429,
                "cached_tokens": 34725,
                "wall_ms": 39941
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "845cbc44b24aae36f61267aca86622b5880c2574ef0eaf0d901f0d13c8e59080",
              "status": "measured",
              "samples": [
                0.62,
                0.67,
                0.65
              ],
              "judge_sample_hashes": [
                "5db569ec9a4acf8227866ccee8dc517284fbd5bb8cb7d2079d0f4399d20b3fc0",
                "4a7c790fc275c05ac670319b96768121fec6b5d362b5253fd123eabe39630aa0",
                "0c5f35a2a52cb361d0ae143a39eb18a251a7e3f5402a1958135d9ac00ccd957f"
              ],
              "mean": 0.646667,
              "stddev": 0.025166,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 18950,
                "output_tokens": 1303,
                "cached_tokens": 18940,
                "wall_ms": 15686
              },
              "judge_usage": {
                "input_tokens": 47106,
                "output_tokens": 1574,
                "cached_tokens": 34659,
                "wall_ms": 32653
              }
            }
          ]
        }
      },
      {
        "id": "severity-labeled-findings",
        "mode": "baseline",
        "outcome": "borderline",
        "score": 0.546667,
        "mean": 0.546667,
        "stddev": 0.189561,
        "samples": [
          0.35,
          0.4,
          0.4
        ],
        "generation_hash": "495c6a858ec85b1a6edc447ebb938c9180afcbac17045584ee268073181eadc9",
        "judge_sample_hashes": [
          "fcfa807405690e2e90977a0293b89f11ccbe48505cab6174705ed8c8b8f421c1",
          "1c0962b3dfb1285f2be599b1578cb27b47ddfb0d85c9517eaae812187df37b07",
          "057834805244a92baa112115c19955b5216316c5651724f016d092d66fa00c0d"
        ],
        "threshold": 0.7,
        "reason": "Secret flagged under a Critical heading, but severity is only coarse section grouping; PII logging demoted to \"Other Issues\", error/promise issues miscalibrated as Critical security, and the cosmetic `data` name Nit omitted entirely.",
        "judge": {
          "model_id": "claude-opus-5",
          "rubric_hash": "a953d281f6a998f2b332172b431e3fd36792516ceefbc87ca00bab9ebcf015b0"
        },
        "judge_usage": {
          "input_tokens": 46230,
          "output_tokens": 2733,
          "cached_tokens": 34075,
          "wall_ms": 47919
        },
        "generation": {
          "n_planned": 3,
          "n_drawn": 4,
          "n_measured": 4,
          "n_unmeasured": 0,
          "stopping_reason": "stabilised",
          "mean": 0.546667,
          "sd": 0.189561,
          "judge_sd_mean": 0.040799,
          "variance_ratio": 4.646217,
          "variance_ratio_unavailable": null,
          "n_truncated": 0,
          "draws": [
            {
              "draw_index": 0,
              "generation_hash": "4549029fb7aaef281f14f563996f6506f5685aee9a3ba168e13d7fb8ddebe09b",
              "status": "measured",
              "samples": [
                0.73,
                0.72,
                0.75
              ],
              "judge_sample_hashes": [
                "1b1f81953742985c31f933db7bd5d97e0b4ea615a65dc9eb125149e463b1b8fe",
                "b24c7b414a52ec1ea091a30bfedb6674e642768bf6db8d0fa8da4c09afb16cf3",
                "33ffb079bf6104a7b14f3af0d5b60c66aca993db914df9986c4ccb143ce68c79"
              ],
              "mean": 0.733333,
              "stddev": 0.015275,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14097,
                "output_tokens": 1158,
                "cached_tokens": 14087,
                "wall_ms": 13660
              },
              "judge_usage": {
                "input_tokens": 46485,
                "output_tokens": 1995,
                "cached_tokens": 34245,
                "wall_ms": 58327
              }
            },
            {
              "draw_index": 1,
              "generation_hash": "7007c3edf6fa6a6845db3e68aa1bc9675bef2846f5bd4027505ba7ea62a78a75",
              "status": "measured",
              "samples": [
                0.4,
                0.4,
                0.35
              ],
              "judge_sample_hashes": [
                "301c383a827d08ef2287d8884ab8ddf677687d36c8a59a1a877f68d4008e009e",
                "ec4fed3e324ba73859883213dc185992dc9fa010c3062cbff32fbb24c9ecc2ec",
                "4a7f065e4894962b1ba185f6aae246d32720aad58761eaf0c1225b0255eee0d7"
              ],
              "mean": 0.383333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14097,
                "output_tokens": 1260,
                "cached_tokens": 14087,
                "wall_ms": 14669
              },
              "judge_usage": {
                "input_tokens": 46581,
                "output_tokens": 2956,
                "cached_tokens": 34309,
                "wall_ms": 58119
              }
            },
            {
              "draw_index": 2,
              "generation_hash": "fd77f33eb6949a84d79cdc746ea4d298fa33d6d3610c6399ac89596d32bff103",
              "status": "measured",
              "samples": [
                0.78,
                0.6,
                0.68
              ],
              "judge_sample_hashes": [
                "52f63c931c2b9d88a3c7be3aba685e424906b163389d711f0997f15676853d10",
                "18fb6988afc75d310142dcfdfdefbe88b2ae66cbfd1ec96d5cb63cf55550a556",
                "04c1b68e91fb7d2466eabb13767f3768f20b5590c2b49d43bfbf536ef61bf2f1"
              ],
              "mean": 0.686667,
              "stddev": 0.090185,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14097,
                "output_tokens": 1153,
                "cached_tokens": 14087,
                "wall_ms": 13181
              },
              "judge_usage": {
                "input_tokens": 46641,
                "output_tokens": 2520,
                "cached_tokens": 34349,
                "wall_ms": 47765
              }
            },
            {
              "draw_index": 3,
              "generation_hash": "495c6a858ec85b1a6edc447ebb938c9180afcbac17045584ee268073181eadc9",
              "status": "measured",
              "samples": [
                0.35,
                0.4,
                0.4
              ],
              "judge_sample_hashes": [
                "fcfa807405690e2e90977a0293b89f11ccbe48505cab6174705ed8c8b8f421c1",
                "1c0962b3dfb1285f2be599b1578cb27b47ddfb0d85c9517eaae812187df37b07",
                "057834805244a92baa112115c19955b5216316c5651724f016d092d66fa00c0d"
              ],
              "mean": 0.383333,
              "stddev": 0.028868,
              "stop_reason": "end_turn",
              "truncated": false,
              "reported_model": "claude-haiku-4-5-20251001",
              "usage": {
                "input_tokens": 14097,
                "output_tokens": 967,
                "cached_tokens": 14087,
                "wall_ms": 11121
              },
              "judge_usage": {
                "input_tokens": 46230,
                "output_tokens": 2733,
                "cached_tokens": 34075,
                "wall_ms": 47919
              }
            }
          ]
        }
      }
    ],
    "aggregates": {
      "with_skill": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 0,
        "mean_score": 0.565834,
        "stddev": null
      },
      "baseline": {
        "case_count": 1,
        "pass_count": 0,
        "borderline_count": 1,
        "mean_score": 0.546667,
        "stddev": null
      },
      "band_rule": "per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases"
    }
  },
  "comparison": {
    "with_skill_score": 0.565834,
    "baseline_score": 0.546667,
    "delta": 0.019167,
    "delta_uncertainty": null,
    "delta_uncertainty_unavailable": "single_case"
  },
  "verification_level": "TESTED",
  "receipt_hash": "3e9152ffbfd278f5bc8ff55da335f14b386e6a51de44694c4298518f1bc3db49",
  "economics": {
    "basis": "metered-equivalent",
    "surface": "claude-cli",
    "with_skill": {
      "call_count": 4,
      "mean_input_tokens": 18950,
      "mean_output_tokens": 1166.5,
      "mean_cost_usd_per_call": 0.024783,
      "median_wall_ms": 14761,
      "wall_ms_p25": 13458,
      "wall_ms_p75": 16007.5,
      "wall_ms_iqr": 2549.5
    },
    "baseline": {
      "call_count": 4,
      "mean_input_tokens": 14097,
      "mean_output_tokens": 1134.5,
      "mean_cost_usd_per_call": 0.01977,
      "median_wall_ms": 13420.5,
      "wall_ms_p25": 12151,
      "wall_ms_p75": 14164.5,
      "wall_ms_iqr": 2013.5
    },
    "skill_incremental_cost_usd_per_call": 0.005013,
    "skill_incremental_cost_usd_per_1k_calls": 5.013,
    "output_tokens_delta": 32,
    "median_wall_ms_delta": 1340.5,
    "judge_excluded": true,
    "judge_overhead": {
      "note": "Measurement overhead imposed by Driftproof, NOT a cost of running the skill. Excluded from every field above.",
      "total_cost_usd": 0.574355,
      "case_rows_measured": 2
    },
    "notes": {
      "absolute_cost": "Absolute per-call cost on a CLI surface includes a fixed harness preamble (observed ~25k input tokens on claude-cli, ~11k on codex) that we do not control. It is identical in the with-skill and baseline arms, so it cancels in the Δ figures — read the incremental columns, not the absolute ones.",
      "cache_pricing": "Cached input tokens are costed at the list input rate (an over-estimate where caching is heavy); cached_tokens is recorded per call so a reader can recompute.",
      "latency": "observed on subscription CLI surface, indicative"
    }
  }
}