{
  "bundle_version": "1",
  "generated_at": "2026-08-28T23:53:47.429Z",
  "batch": {
    "id": "sample-evidence-pack",
    "ref": "BAT-0001",
    "kind": "INDEPENDENT_QA",
    "item_type_code": "RUBRIC_SCORED",
    "replication_factor": 2,
    "gold_target_pct": null,
    "status": "COMPLETED",
    "created_at": "2026-08-28T23:53:45.953Z"
  },
  "sampling": null,
  "judgments": [
    {
      "item_id": "item-0001",
      "external_ref": "helpsteer2-0",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-1",
      "original_label": "PASS",
      "final_label": "PASS",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 2.8
    },
    {
      "item_id": "item-0002",
      "external_ref": "helpsteer2-1",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-2",
      "original_label": "PASS",
      "final_label": "PASS",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 3.6
    },
    {
      "item_id": "item-0003",
      "external_ref": "helpsteer2-2",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-3",
      "original_label": "PASS",
      "final_label": "PASS",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 2.6
    },
    {
      "item_id": "item-0004",
      "external_ref": "helpsteer2-3",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-4",
      "original_label": "PASS",
      "final_label": "PASS",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 2.4
    },
    {
      "item_id": "item-0005",
      "external_ref": "helpsteer2-4",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-5",
      "original_label": "PASS",
      "final_label": "PASS",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 3
    },
    {
      "item_id": "item-0006",
      "external_ref": "helpsteer2-5",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-6",
      "original_label": "FAIL",
      "final_label": "FAIL",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 1.4
    },
    {
      "item_id": "item-0007",
      "external_ref": "helpsteer2-6",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-1",
      "original_label": "PASS",
      "final_label": "PASS",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 3.2
    },
    {
      "item_id": "item-0008",
      "external_ref": "helpsteer2-7",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-2",
      "original_label": "PASS",
      "final_label": "PASS",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 3.6
    },
    {
      "item_id": "item-0009",
      "external_ref": "helpsteer2-8",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-3",
      "original_label": "PASS",
      "final_label": "PASS",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 2.4
    },
    {
      "item_id": "item-0010",
      "external_ref": "helpsteer2-9",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-4",
      "original_label": "PASS",
      "final_label": "PASS",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 3.2
    },
    {
      "item_id": "item-0011",
      "external_ref": "helpsteer2-10",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-5",
      "original_label": "PASS",
      "final_label": "PASS",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 2.8
    },
    {
      "item_id": "item-0012",
      "external_ref": "helpsteer2-11",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-6",
      "original_label": "FAIL",
      "final_label": "FAIL",
      "is_adjudication": true,
      "gold_correct": null,
      "rubric_overall_score": 1.2
    },
    {
      "item_id": "item-0013",
      "external_ref": "helpsteer2-12",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-1",
      "original_label": "FAIL",
      "final_label": "FAIL",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 1.6
    },
    {
      "item_id": "item-0014",
      "external_ref": "helpsteer2-13",
      "kind": "STANDARD",
      "original_contributor_key": "helpsteer2-worker-2",
      "original_label": "PASS",
      "final_label": "PASS",
      "is_adjudication": false,
      "gold_correct": null,
      "rubric_overall_score": 2.8
    },
    {
      "item_id": "item-0015",
      "external_ref": "gold-pass-1",
      "kind": "GOLD",
      "original_contributor_key": null,
      "original_label": null,
      "final_label": "PASS",
      "is_adjudication": false,
      "gold_correct": true,
      "rubric_overall_score": 4
    },
    {
      "item_id": "item-0016",
      "external_ref": "gold-fail-1",
      "kind": "GOLD",
      "original_contributor_key": null,
      "original_label": null,
      "final_label": "FAIL",
      "is_adjudication": false,
      "gold_correct": true,
      "rubric_overall_score": 0
    }
  ],
  "qa_responses": [
    {
      "item_id": "item-0001",
      "responder_ref": "reviewer-a",
      "label": "PASS"
    },
    {
      "item_id": "item-0001",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0002",
      "responder_ref": "reviewer-a",
      "label": "PASS"
    },
    {
      "item_id": "item-0002",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0003",
      "responder_ref": "reviewer-a",
      "label": "PASS"
    },
    {
      "item_id": "item-0003",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0004",
      "responder_ref": "reviewer-a",
      "label": "PASS"
    },
    {
      "item_id": "item-0004",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0005",
      "responder_ref": "reviewer-a",
      "label": "PASS"
    },
    {
      "item_id": "item-0005",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0006",
      "responder_ref": "reviewer-a",
      "label": "FAIL"
    },
    {
      "item_id": "item-0006",
      "responder_ref": "reviewer-b",
      "label": "FAIL"
    },
    {
      "item_id": "item-0007",
      "responder_ref": "reviewer-a",
      "label": "PASS"
    },
    {
      "item_id": "item-0007",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0008",
      "responder_ref": "reviewer-a",
      "label": "PASS"
    },
    {
      "item_id": "item-0008",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0009",
      "responder_ref": "reviewer-a",
      "label": "PASS"
    },
    {
      "item_id": "item-0009",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0010",
      "responder_ref": "reviewer-a",
      "label": "PASS"
    },
    {
      "item_id": "item-0010",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0011",
      "responder_ref": "reviewer-a",
      "label": "PASS"
    },
    {
      "item_id": "item-0011",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0012",
      "responder_ref": "reviewer-a",
      "label": "FAIL"
    },
    {
      "item_id": "item-0012",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0013",
      "responder_ref": "reviewer-a",
      "label": "FAIL"
    },
    {
      "item_id": "item-0013",
      "responder_ref": "reviewer-b",
      "label": "FAIL"
    },
    {
      "item_id": "item-0014",
      "responder_ref": "reviewer-a",
      "label": "PASS"
    },
    {
      "item_id": "item-0014",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0015",
      "responder_ref": "reviewer-a",
      "label": "PASS"
    },
    {
      "item_id": "item-0015",
      "responder_ref": "reviewer-b",
      "label": "PASS"
    },
    {
      "item_id": "item-0016",
      "responder_ref": "reviewer-a",
      "label": "FAIL"
    },
    {
      "item_id": "item-0016",
      "responder_ref": "reviewer-b",
      "label": "FAIL"
    }
  ],
  "defect_records": [
    {
      "item_id": "item-0012",
      "defect_code": "FACTUAL_ERROR",
      "severity": "MAJOR"
    }
  ],
  "rubric": {
    "criteria": [
      {
        "key": "helpfulness",
        "prompt": "How helpful is the response?",
        "weight": 1,
        "scale_max": 4,
        "scale_min": 0
      },
      {
        "key": "correctness",
        "prompt": "How correct/factual is the response?",
        "weight": 1,
        "scale_max": 4,
        "scale_min": 0
      },
      {
        "key": "coherence",
        "prompt": "How coherent is the response?",
        "weight": 1,
        "scale_max": 4,
        "scale_min": 0
      },
      {
        "key": "complexity",
        "prompt": "How complex is the response?",
        "weight": 1,
        "scale_max": 4,
        "scale_min": 0
      },
      {
        "key": "verbosity",
        "prompt": "How verbose is the response?",
        "weight": 1,
        "scale_max": 4,
        "scale_min": 0
      }
    ],
    "overall_pass_threshold": 2
  },
  "metrics": {
    "id": "sample-evidence-pack-metrics",
    "sample_n": 16,
    "population_n": 16,
    "agreement_value": 1,
    "agreement_method": "COHENS_KAPPA",
    "gold_accuracy_pct": 100,
    "rubric_pass_rate_pct": 78.57142857142857,
    "defect_rate_pct": 6.25,
    "breakdowns": {
      "defect_breakdown": {
        "FACTUAL_ERROR": 1
      },
      "supplementary_agreement": {
        "gwet_ac1": 1
      },
      "defect_severity_breakdown": {
        "MAJOR": 1
      }
    },
    "per_contributor": {
      "helpsteer2-worker-1": {
        "items_sampled": 3,
        "reviewer_agreement": 1,
        "gold_set_accuracy_pct": null
      },
      "helpsteer2-worker-2": {
        "items_sampled": 3,
        "reviewer_agreement": 1,
        "gold_set_accuracy_pct": null
      },
      "helpsteer2-worker-3": {
        "items_sampled": 2,
        "reviewer_agreement": 1,
        "gold_set_accuracy_pct": null
      },
      "helpsteer2-worker-4": {
        "items_sampled": 2,
        "reviewer_agreement": 1,
        "gold_set_accuracy_pct": null
      },
      "helpsteer2-worker-5": {
        "items_sampled": 2,
        "reviewer_agreement": 1,
        "gold_set_accuracy_pct": null
      },
      "helpsteer2-worker-6": {
        "items_sampled": 2,
        "reviewer_agreement": 1,
        "gold_set_accuracy_pct": null
      }
    },
    "effort": {
      "items_judged": 16,
      "median_item_seconds": 0,
      "total_active_seconds": 0
    },
    "intervals": {
      "gwet_ac1": {
        "hi": 1,
        "lo": 1,
        "seed": "bootstrap:ae67994d72ad72ab438a535674097d31c8c146567d707de43273b5d5a04f9201:gwet_ac1",
        "level": 0.95,
        "point": 1,
        "method": "BOOTSTRAP_PERCENTILE",
        "resamples": 2000
      },
      "agreement": {
        "hi": 1,
        "lo": 1,
        "seed": "bootstrap:ae67994d72ad72ab438a535674097d31c8c146567d707de43273b5d5a04f9201:cohens_kappa",
        "level": 0.95,
        "point": 1,
        "method": "BOOTSTRAP_PERCENTILE",
        "resamples": 2000
      },
      "defect_rate": {
        "wilson": {
          "hi": 28.328737570298944,
          "lo": 1.1119344764642518,
          "level": 0.95,
          "point": 6.25,
          "method": "WILSON_SCORE"
        }
      },
      "gold_accuracy": {
        "wilson": {
          "hi": 100,
          "lo": 34.2380227506653,
          "level": 0.95,
          "point": 100,
          "method": "WILSON_SCORE"
        },
        "bootstrap": {
          "hi": 100,
          "lo": 100,
          "seed": "bootstrap:ae67994d72ad72ab438a535674097d31c8c146567d707de43273b5d5a04f9201:gold_accuracy",
          "level": 0.95,
          "point": 100,
          "method": "BOOTSTRAP_PERCENTILE",
          "resamples": 2000
        }
      },
      "rubric_pass_rate": {
        "wilson": {
          "hi": 92.42861328730683,
          "lo": 52.410769413399706,
          "level": 0.95,
          "point": 78.57142857142857,
          "method": "WILSON_SCORE"
        },
        "bootstrap": {
          "hi": 100,
          "lo": 57.14285714285714,
          "seed": "bootstrap:ae67994d72ad72ab438a535674097d31c8c146567d707de43273b5d5a04f9201:rubric_pass_rate",
          "level": 0.95,
          "point": 78.57142857142857,
          "method": "BOOTSTRAP_PERCENTILE",
          "resamples": 2000
        }
      }
    },
    "validator_agreement": {
      "value": 0.8181818181818182,
      "method": "COHENS_KAPPA",
      "n_items": 16,
      "interval": {
        "hi": 1,
        "lo": 0.2544186046511653,
        "seed": "bootstrap:ae67994d72ad72ab438a535674097d31c8c146567d707de43273b5d5a04f9201:validator_agreement:cohens_kappa",
        "level": 0.95,
        "point": 0.8181818181818182,
        "method": "BOOTSTRAP_PERCENTILE",
        "resamples": 2000
      },
      "n_validators": 2
    },
    "inputs_hash": "ae67994d72ad72ab438a535674097d31c8c146567d707de43273b5d5a04f9201",
    "computed_by": null,
    "created_at": "2026-08-28T23:53:47.480Z"
  },
  "audit_excerpt": [],
  "recompute": {
    "notes": "Recompute every field in `metrics` from `judgments` + `defect_records` using the `recompute.statistics` definitions below, and the `recompute.bootstrap`/`recompute.wilson` procedures for the intervals in `metrics.intervals` (each interval already records its own `seed` — feed it into `recompute.rng` verbatim, never re-derive it). `metrics.inputs_hash` is a SHA-256 of the canonical (deterministically key-sorted) JSON of the full original judgment RESPONSE objects — not the labels shipped in `judgments` here — so it is reported as provenance only and is not independently re-derivable from this bundle; verification is the recomputed NUMBERS matching, not the hash.",
    "rng": {
      "name": "mulberry32-sha256",
      "seed_derivation": "state0 = first 4 bytes of SHA-256(seed_string, utf-8), interpreted as a little-endian uint32.",
      "generator": "mulberry32(state), all arithmetic on 32-bit values (mask with 0xFFFFFFFF after every step; shifts are LOGICAL/unsigned): state = (state + 0x6D2B79F5) mod 2^32; t = state; t = ((t ^ (t >> 15)) * (t | 1)) mod 2^32 [Math.imul-equivalent: exact multiply, keep low 32 bits]; t = t ^ ((t + (((t ^ (t >> 7)) * (t | 61)) mod 2^32)) mod 2^32) [XOR, not addition — the sum inside is computed then reduced mod 2^32 before the XOR]; return (t ^ (t >> 14)) mod 2^32 / 2^32. Called once per draw; returns a float in [0, 1)."
    },
    "bootstrap": {
      "method": "BOOTSTRAP_PERCENTILE",
      "procedure": "For each of `resamples` iterations: draw n indices in [0, n) WITH replacement (index_i = floor(rng() * n) for i in 0..n-1); compute the statistic over the resampled array (indices applied in draw order); collect the resample statistics, sort ascending.",
      "percentile_method": "Linear interpolation (numpy default / R type-7): given sorted values and probability p, idx = p * (len(values) - 1); lo = floor(idx); hi = ceil(idx); frac = idx - lo; result = values[lo] * (1 - frac) + values[hi] * frac. lo/hi = alpha/2 and 1-alpha/2 for a `level`-confidence interval (alpha = 1 - level)."
    },
    "wilson": {
      "method": "WILSON_SCORE",
      "formula": "z = 1.959963984540054 for level=0.95 (the only level this build emits); p = successes/n; center = (p + z^2/(2n)) / (1 + z^2/n); margin = z * sqrt(p(1-p)/n + z^2/(4n^2)) / (1 + z^2/n); interval = [max(0, center - margin), min(1, center + margin)]."
    },
    "statistics": {
      "percent_agreement": "Fraction of judgment pairs (original_label, final_label) — both non-null — that agree. Units with a null original_label are excluded.",
      "cohens_kappa": "(po - pe) / (1 - pe), po = observed agreement, pe = sum over categories of (marginal_a[c]/n) * (marginal_b[c]/n), over the same non-null pairs. pe==1 -> kappa=1 (degenerate, not 0/0).",
      "krippendorff_alpha": "Nominal, coincidence-matrix method over the 2-rater (original_label, final_label) pairs, treating a null original_label as a missing rating for that unit (units need >=2 non-null ratings to contribute). alpha = 1 - (n-1)(n-sum_diag) / (n^2 - sum(marginal_c^2)); degenerate denominator -> alpha=1.",
      "gwet_ac1": "Same po as Cohen's kappa; pe_AC1 = (1/(q-1)) * sum_c[pi_c*(1-pi_c)], pi_c = pooled prevalence of category c across BOTH raters / (2n), q = categories in use. AC1 = (po - pe_AC1)/(1 - pe_AC1). Computed only over pairs with a non-null original_label. NOMINAL-label-kind item types only (breakdowns.supplementary_agreement.gwet_ac1) — null/omitted for ordinal types, which report gwet_ac2 instead.",
      "gold_accuracy": "100 * (count of gold_correct==true) / (count of kind=GOLD judgments with gold_correct != null).",
      "defect_rate": "100 * (count of items with >=1 defect_record whose severity is MAJOR or CRITICAL) / (count of judgments, incl. GOLD).",
      "rubric_pass_rate": "RUBRIC_SCORED only. 100 * (count of non-GOLD judgments with rubric_overall_score >= rubric.overall_pass_threshold) / (count of non-GOLD judgments with a non-null rubric_overall_score).",
      "weighted_kappa": "B4, ORDINAL-label-kind item types only (CONVERSATION_TURN, ABSOLUTE_SCALE), selected as agreement_method=\"WEIGHTED_KAPPA_QUADRATIC\" (2 raters, no missing original_label). original_label/final_label are parsed as numbers; categories = the sorted set of distinct numeric values actually observed across both. Build a q x q count matrix O over (original,final) pairs and its independence-expected counterpart E[i][j] = rowSum[i]*colSum[j]/n (i,j = each value's 0-indexed RANK among categories, ascending). quadratic weight w[i][j] = (i-j)^2. kappa_w = 1 - sum(w*O) / sum(w*E). (Source: scikit-learn's cohen_kappa_score, sklearn/metrics/_classification.py.)",
      "krippendorff_alpha_ordinal": "B4, ORDINAL-label-kind item types, selected instead of weighted_kappa when there are missing original_label values. Same coincidence-matrix machinery as krippendorff_alpha above, but categories are ranked numerically and the nominal delta2(c,k)=[c!=k] is replaced by delta2_ordinal(c,k) = (sum_{g=c}^{k} n_g - (n_c+n_k)/2)^2, summed over categories g in RANK order from c to k inclusive (n_g = that category's marginal). alpha = 1 - (n-1)*sum_{c!=k}[o_ck*delta2(c,k)] / sum_{c!=k}[n_c*n_k*delta2(c,k)]. (Source: Krippendorff, K. (2011), \"Computing Krippendorff's Alpha-Reliability\", pp.6,8.)",
      "gwet_ac2": "B4, ORDINAL-label-kind item types only (breakdowns.supplementary_agreement.gwet_ac2) — supplementary, never selected as agreement_method. Same pooled-prevalence pi_c as gwet_ac1, but pa is a WEIGHTED mean agreement (pa = mean_i W[rank(original_i)][rank(final_i)]) and pe = sum(W) * sum_c[pi_c*(1-pi_c)] / (q*(q-1)), using the AGREEMENT weight matrix W[k][l] = 1 - (categ_k-categ_l)^2/(max-min)^2 (quadratic; categ values are the same numeric categories weighted_kappa uses, not ranks). AC2 = (pa-pe)/(1-pe). (Source: Gwet, K.L., irrCAC R package, agree.coeff3.raw.r + weights.gen.r, specialized to 2 raters.)"
    },
    "validator_agreement": "Group `qa_responses` by item_id; an item QUALIFIES only if it has >=2 responses with a non-null label (drop the rest of that item's responses first). null (never fabricated) if no item qualifies. Let V = the set of distinct responder_ref across every qualifying item. If |V| == 2 AND every qualifying item has EXACTLY those same 2 responder_ref values (never more, never a different pair) -> method=COHENS_KAPPA: sort V ascending, \"rater A\"/\"rater B\" = V[0]/V[1], build paired arrays in qualifying-item order, kappa = the cohens_kappa formula above, its bootstrap CI seeded `${base_seed}:validator_agreement:cohens_kappa`, n_validators=2. Otherwise -> method=KRIPPENDORFF_ALPHA: units = one array per qualifying item of that item's labels (unpaired, any count/identity), alpha = the krippendorff_alpha formula above, its bootstrap CI seeded `${base_seed}:validator_agreement:krippendorff_alpha`, n_validators=|V|. n_items = count of qualifying items either way. `base_seed` is the same `bootstrap:${inputs_hash}` prefix every other interval in `intervals` derives from.",
    "risk_sampling": {
      "allocation": "Group population_item_ids by each item's original_contributor_key (judgments carry it) into strata, \"(none)\" for a null key. Beta(1,19)-smoothed rate per stratum: r_hat_k = (D_k + a) / (M_k + a + b), where M_k/D_k are contributor_stats.items_completed/defects_involved for that stratum key (0/0 if no stats row exists). Pooled rate r_bar = (sum of D_k over KNOWN strata + a) / (sum of M_k over KNOWN strata + a + b). multiplier_k = known ? clamp(r_hat_k / r_bar, clamp[0], clamp[1]) : unknown_multiplier (2.0 — unknown is treated as RISKY, not neutral). min_k = max(1, ceil(floor_pct * N_k)); residual = n - sum(min_k); weighted largest-remainder (Hamilton) over weight_k = N_k * multiplier_k allocates the residual, capped at N_k - min_k per stratum, re-pooling any stratum that would exceed its cap and repeating until none do; n_k = min_k + alloc_k. Per-stratum draw: sampleRandom(stratum_ids_in_population_order, n_k, `${seed}:${key}`) — same mulberry32-over-sha256 generator as `rng` above.",
      "multiplier": "Recorded verbatim per stratum in `sampling.strata[].multiplier`/`.smoothed_rate`/`.known` — recompute and compare directly rather than re-deriving `a`/`b`/`clamp`/`unknown_multiplier`, which travel in `sampling.params`.",
      "horvitz_thompson": "Within-stratum SRS makes inclusion probability constant per stratum, reducing HT to the stratified form. Group `judgments` (JUDGED sampled items only — i.e. every row in this bundle) by original_contributor_key; d_k = count with >=1 defect_records row of severity MAJOR/CRITICAL; n'_k = count of judgments in that stratum. r_k = n'_k > 0 ? d_k/n'_k : null (a stratum with n'_k=0 is DROPPED, contributing nothing). N_eff = sum of N_k (`sampling.strata[].populationSize`) over non-dropped strata. defect_rate_HT_pct = 100 * sum_k (N_k/N_eff) * r_k. This value is `metrics.defect_rate_pct` when `sampling.method === \"RISK_WEIGHTED_BY_CONTRIBUTOR\"`; the naive (unweighted, `recompute.statistics.defect_rate`) value moves to `metrics.breakdowns.defect_rate_naive_pct` instead.",
      "stratified_bootstrap": "Resample WITHIN each non-dropped stratum only, weights FIXED at N_k/N_eff (the same weights the point estimate used) — never resample across strata. One persistent rng per stratum, seeded `${seed}:stratum:${key}` where seed is `bootstrap:${inputs_hash}:defect_rate_ht`, created once and advanced across every resample (never reseeded per resample). Each resample draws n'_k indices in [0, n'_k) WITH replacement per stratum, averages that stratum's 0/1 defect flags over the draw, and sums weight_k * mean_k across strata; sort the `resamples` draws ascending, take the alpha/2 and 1-alpha/2 percentiles (same linear-interpolation method as `bootstrap.percentile_method`). Refuses (null) when EVERY stratum has fewer than 2 judged items."
    }
  }
}
