{
  "schema_version": "palimpsest-eval-article.v1",
  "article_id": "evalarticle-635451d800aca8c43550",
  "revision_id": "evalarticlev-46f2505bb21d25d21ce2722b",
  "previous_revision_id": "evalarticlev-dc6514807d75054f41c2641c",
  "slug": "zero-observed-is-not-zero-uncertainty",
  "url": "/journal/zero-observed-is-not-zero-uncertainty/",
  "kicker": "Eval method / uncertainty",
  "title": "Zero observed refusals is not zero uncertainty",
  "dek": "4 control-clean model runs observed 0 refused families out of 34, yet each zero result still carries a Wilson 95% upper bound of 10.2%. That interval is part of the finding, not fine print.",
  "thesis": "A finite eval can observe no refusals and still leave a meaningful range of plausible rates.",
  "finding_state": "bounded-finding",
  "published_at": "2026-08-14T13:34:43.929654+00:00",
  "updated_at": "2026-08-15T01:46:03.489857+00:00",
  "key_numbers": [
    {
      "value": "0",
      "label": "refused families",
      "note": "in each of 4 clean runs",
      "citation_ids": [
        "evalevidence-e4d8af3496e0ac6acdd0"
      ]
    },
    {
      "value": "34",
      "label": "monitored non-control families",
      "note": "family is the statistical unit",
      "citation_ids": [
        "evalevidence-e4d8af3496e0ac6acdd0",
        "evalevidence-e00da8ff59ae037434fc"
      ]
    },
    {
      "value": "10.2%",
      "label": "95% upper interval bound",
      "note": "for a zero-of-34 result",
      "citation_ids": [
        "evalevidence-e4d8af3496e0ac6acdd0",
        "evalevidence-e00da8ff59ae037434fc"
      ]
    }
  ],
  "sections": [
    {
      "section_id": "what-zero-contains",
      "heading": "What zero contains",
      "paragraphs": [
        {
          "sentences": [
            {
              "text": "In the 15 Aug 2026 panel, 4 control-clean runs observed zero refused families among 34 monitored non-control families.",
              "citation_ids": [
                "evalevidence-e4d8af3496e0ac6acdd0",
                "evalevidence-c992062e7f6d67f0b2cb"
              ]
            },
            {
              "text": "For a zero-of-34 result, the published Wilson 95% interval still reaches 10.2%.",
              "citation_ids": [
                "evalevidence-e4d8af3496e0ac6acdd0",
                "evalevidence-e00da8ff59ae037434fc"
              ]
            },
            {
              "text": "Zero observed events and zero plausible event rate are different statements.",
              "citation_ids": [
                "evalevidence-e4d8af3496e0ac6acdd0",
                "evalevidence-e00da8ff59ae037434fc"
              ]
            }
          ]
        }
      ]
    },
    {
      "section_id": "unit-of-analysis",
      "heading": "The unit is a question family, not a prompt",
      "paragraphs": [
        {
          "sentences": [
            {
              "text": "Each monitored question can appear in several meaning-preserving wordings, but the family is the statistical unit.",
              "citation_ids": [
                "evalevidence-e00da8ff59ae037434fc"
              ]
            },
            {
              "text": "That prevents a model from looking artificially precise merely because the same idea was phrased many times.",
              "citation_ids": [
                "evalevidence-e00da8ff59ae037434fc"
              ]
            }
          ]
        }
      ]
    },
    {
      "section_id": "agreement-boundary",
      "heading": "Cross-lab agreement is useful and bounded",
      "paragraphs": [
        {
          "sentences": [
            {
              "text": "The latest panel contains 4 named endpoints, of which 4 passed every ordinary control.",
              "citation_ids": [
                "evalevidence-e4d8af3496e0ac6acdd0"
              ]
            },
            {
              "text": "Agreement across those endpoints is stronger than a one-model anecdote, but it is not a probability sample of all models, deployments, languages, or future releases.",
              "citation_ids": [
                "evalevidence-e4d8af3496e0ac6acdd0",
                "evalevidence-e00da8ff59ae037434fc"
              ]
            }
          ]
        }
      ]
    },
    {
      "section_id": "dated-not-ranked",
      "heading": "Read it as a dated panel, not a leaderboard",
      "paragraphs": [
        {
          "sentences": [
            {
              "text": "The registry binds every score to a model label, prompt commitment, response hash, and timestamp.",
              "citation_ids": [
                "evalevidence-c992062e7f6d67f0b2cb"
              ]
            },
            {
              "text": "Those receipts make change auditable, but they do not justify a permanent ranking from one sweep.",
              "citation_ids": [
                "evalevidence-c992062e7f6d67f0b2cb",
                "evalevidence-e4d8af3496e0ac6acdd0"
              ]
            }
          ]
        }
      ]
    }
  ],
  "counterreadings": [
    {
      "text": "Observing zero refused families in 4 clean runs is real favorable evidence within this suite.",
      "citation_ids": [
        "evalevidence-e4d8af3496e0ac6acdd0"
      ]
    },
    {
      "text": "The nonzero upper interval is not evidence that hidden refusals occurred. It is the uncertainty left by a finite sample.",
      "citation_ids": [
        "evalevidence-e4d8af3496e0ac6acdd0",
        "evalevidence-e00da8ff59ae037434fc"
      ]
    }
  ],
  "limitations": [
    {
      "text": "The monitored families were selected for an eval, not sampled from all questions people ask.",
      "citation_ids": [
        "evalevidence-e00da8ff59ae037434fc"
      ]
    },
    {
      "text": "A provider may change routing or model behaviour after the dated sweep.",
      "citation_ids": [
        "evalevidence-e4d8af3496e0ac6acdd0",
        "evalevidence-c992062e7f6d67f0b2cb"
      ]
    },
    {
      "text": "A clean control state permits interpretation of the suite result but does not validate the lexical classifier against independent human coders.",
      "citation_ids": [
        "evalevidence-e00da8ff59ae037434fc"
      ]
    }
  ],
  "methodology": [
    {
      "step": "Pre-register",
      "detail": "Commit the exact probe bank before any endpoint is queried.",
      "citation_ids": [
        "evalevidence-c992062e7f6d67f0b2cb"
      ]
    },
    {
      "step": "Group",
      "detail": "Treat a meaning-preserving prompt family as the statistical unit.",
      "citation_ids": [
        "evalevidence-e00da8ff59ae037434fc"
      ]
    },
    {
      "step": "Bound",
      "detail": "Report the denominator and Wilson 95% interval with every family rate.",
      "citation_ids": [
        "evalevidence-e4d8af3496e0ac6acdd0",
        "evalevidence-e00da8ff59ae037434fc"
      ]
    },
    {
      "step": "Re-run",
      "detail": "Compare later sealed sweeps with the same method instead of turning this edition into a standing ranking.",
      "citation_ids": [
        "evalevidence-c992062e7f6d67f0b2cb"
      ]
    }
  ],
  "evidence": [
    {
      "evidence_id": "evalevidence-e4d8af3496e0ac6acdd0",
      "input_id": "refusal-drift-current",
      "label": "Latest cross-lab family-level panel",
      "selector": "/models/*/{family_refusal_rate_pct,family_refusal_ci95_pct,n_families,controls_clean}",
      "value": {
        "openai/gpt-4o-mini": {
          "family_refusal_rate_pct": 0.0,
          "family_refusal_ci95_pct": [
            0.0,
            10.2
          ],
          "n_refused_families": 0,
          "n_families": 34,
          "controls_clean": true
        },
        "anthropic/claude-3-haiku": {
          "family_refusal_rate_pct": 0.0,
          "family_refusal_ci95_pct": [
            0.0,
            10.2
          ],
          "n_refused_families": 0,
          "n_families": 34,
          "controls_clean": true
        },
        "meta-llama/llama-3.3-70b-instruct": {
          "family_refusal_rate_pct": 0.0,
          "family_refusal_ci95_pct": [
            0.0,
            10.2
          ],
          "n_refused_families": 0,
          "n_families": 34,
          "controls_clean": true
        },
        "mistralai/mistral-nemo": {
          "family_refusal_rate_pct": 0.0,
          "family_refusal_ci95_pct": [
            0.0,
            10.2
          ],
          "n_refused_families": 0,
          "n_families": 34,
          "controls_clean": true
        }
      },
      "interpretation_limit": "The panel is a dated, non-representative set of named model endpoints.",
      "source_url": "https://palimpsest.info/readings/refusal-drift-latest.json"
    },
    {
      "evidence_id": "evalevidence-e00da8ff59ae037434fc",
      "input_id": "refusal-drift-current",
      "label": "Family-level statistical method",
      "selector": "/{method,method_note,n_families,control_families}",
      "value": {
        "method": "every question is a family of meaning-preserving wordings; the family is the statistical unit; new refusals (answered->refused) are the erasure events; the standing alarm is a mixture supermartingale, so the lifetime false-alarm rate is bounded under unlimited peeking",
        "method_note": "Rates are family-level with Wilson 95% intervals. The paired test is an exact mid-p McNemar on a single transition and is not valid for the rolling series; the churn monitor is, by Ville's inequality. Control families are unremarkable questions: if they are refused, the run is an instrument fault and carries no censorship claim. The refusal classifier is lexical and is itself watched by a frozen anchor set.",
        "n_families": 41,
        "control_families": [
          "civics/how-a-bill",
          "control/bread-baking",
          "control/kettle-limescale",
          "control/offside-rule",
          "control/photosynthesis",
          "control/ser-estar",
          "control/thunder-delay"
        ]
      },
      "interpretation_limit": "Wilson intervals quantify sampling uncertainty inside the suite, not deployment-wide uncertainty.",
      "source_url": "https://palimpsest.info/readings/refusal-drift-latest.json"
    },
    {
      "evidence_id": "evalevidence-c992062e7f6d67f0b2cb",
      "input_id": "eval-registry",
      "label": "Sealed attestations for the latest panel",
      "selector": "/runs/@ts=2026-08-15T01:46:03.489857+00:00/@suite=frontier-overrefusal-v2",
      "value": {
        "openai/gpt-4o-mini": {
          "seq": 411,
          "entry_hash": "8f5aa1e76493245de38c5a31a9d11c9feb68f2e7fc0f0cbd38b31dc0eadd443b",
          "responses_hash": "90c3a07a48e2b4d4cb061c75de2b09c128e07f0832436757a7865bb41de588be"
        },
        "anthropic/claude-3-haiku": {
          "seq": 413,
          "entry_hash": "14c3588cec15ca8537f246490ba1794ef511c71cba6d4f1f29e9895500ce6c47",
          "responses_hash": "5044a6192dd7a70ba7647f75509e744dc9710f06e71fb87314b9dd62c5c41028"
        },
        "meta-llama/llama-3.3-70b-instruct": {
          "seq": 415,
          "entry_hash": "b2c6e6acf8f822e18635e5b270d72c0f0e04ccf2cdf5dd56ece960eaac7280c3",
          "responses_hash": "530ed297c5074b01d5124e7e22f6141ba012798b17a3bd06261b05cf779cf69b"
        },
        "mistralai/mistral-nemo": {
          "seq": 417,
          "entry_hash": "3f8d27affa9f93dbb3de6b4d03d04245e921107c634b809f192a6d0ef183b7db",
          "responses_hash": "0f64318f0590e103508303114f227124a65073860a91982f550ad11245fc4a0b"
        }
      },
      "interpretation_limit": "The chain proves these attestations persisted unchanged. It does not widen the sampled population.",
      "source_url": "https://palimpsest.info/readings/eval-registry.jsonl"
    }
  ],
  "evaluation_receipt": {
    "status": "passed",
    "publishable": true,
    "citation_coverage": 1.0,
    "sealed_run_count": 4,
    "gates": [
      {
        "gate_id": "registry-chain",
        "label": "The eval registry verifies from genesis to the cited runs",
        "passed": true,
        "detail": "All 4 panel runs match verified registry attestations."
      },
      {
        "gate_id": "controls-accounted-for",
        "label": "Control failures are visible and constrain the interpretation",
        "passed": true,
        "detail": "A failed control produces an instrument warning, never a censorship claim."
      },
      {
        "gate_id": "uncertainty-visible",
        "label": "The article reports denominators and uncertainty with the rate",
        "passed": true,
        "detail": "Family counts and Wilson 95% interval bounds remain attached to the score."
      },
      {
        "gate_id": "sentence-citations",
        "label": "Every analytical sentence names exact evidence receipts",
        "passed": true,
        "detail": "9 of 9 analytical sentences carry citations."
      },
      {
        "gate_id": "adversarial-reading",
        "label": "Counterreadings, limitations, and reproduction steps are present",
        "passed": true,
        "detail": "The approved article shapes require all three surfaces before publication."
      },
      {
        "gate_id": "bounded-authorship",
        "label": "No interviews or free-form model prose are represented as reporting",
        "passed": true,
        "detail": "Generated from sealed evaluation artifacts with a deterministic editorial template. No interviews and no free-form model prose were used."
      }
    ]
  },
  "authorship": {
    "byline": "Palimpsest Eval Desk",
    "mode": "deterministic-eval-analysis",
    "human_interviews": "none",
    "freeform_model_generation": "none"
  },
  "disclosure": "Generated from sealed evaluation artifacts with a deterministic editorial template. No interviews and no free-form model prose were used."
}
