{
  "schema_version": "palimpsest-eval-article.v1",
  "article_id": "evalarticle-7e2fce8bb9fd3a94cec7",
  "revision_id": "evalarticlev-d76b3f2a015f3f16c467476b",
  "previous_revision_id": "evalarticlev-8e50050891e113d43417b344",
  "slug": "what-the-eval-registry-can-prove-today",
  "url": "/journal/what-the-eval-registry-can-prove-today/",
  "kicker": "Eval integrity / registry status",
  "title": "The eval registry verifies 540 attestations end to end",
  "dek": "The current served chain contains 6 preregistrations and 534 runs across 7 model endpoints. Its hashes make revision detectable; they do not make the eval construct valid by themselves.",
  "thesis": "Tamper evidence answers whether the served eval record changed, not whether the underlying measurement deserves a broader claim.",
  "finding_state": "bounded-finding",
  "published_at": "2026-08-15T12:55:00.684023+00:00",
  "updated_at": "2026-08-23T13:01:08.138128+00:00",
  "key_numbers": [
    {
      "value": "540",
      "label": "verified attestations",
      "note": "head sequence 539",
      "citation_ids": [
        "evalevidence-2ec4f8b64f3c3c75e13a"
      ]
    },
    {
      "value": "6",
      "label": "preregistrations",
      "note": "probe commitments recorded before linked runs",
      "citation_ids": [
        "evalevidence-2ec4f8b64f3c3c75e13a"
      ]
    },
    {
      "value": "534",
      "label": "sealed runs",
      "note": "across 7 named endpoints",
      "citation_ids": [
        "evalevidence-2ec4f8b64f3c3c75e13a"
      ]
    }
  ],
  "sections": [
    {
      "section_id": "chain-verdict",
      "heading": "What verification establishes",
      "paragraphs": [
        {
          "sentences": [
            {
              "text": "The served registry verifies from genesis through 540 attestations, ending at sequence 539.",
              "citation_ids": [
                "evalevidence-2ec4f8b64f3c3c75e13a"
              ]
            },
            {
              "text": "It contains 6 preregistrations and 534 runs across 7 named model endpoints.",
              "citation_ids": [
                "evalevidence-2ec4f8b64f3c3c75e13a"
              ]
            }
          ]
        }
      ]
    },
    {
      "section_id": "current-edition",
      "heading": "What entered the current edition",
      "paragraphs": [
        {
          "sentences": [
            {
              "text": "The latest refusal reading matches 4 registry runs at 2026-08-23T13:01:08.138128+00:00 with no missing panel endpoint.",
              "citation_ids": [
                "evalevidence-bba92a121d5e5d283024",
                "evalevidence-80711deb0314c9522bd2"
              ]
            },
            {
              "text": "Each matched run binds its probe commitment, response digest, model label, metrics, and predecessor hash into the chain.",
              "citation_ids": [
                "evalevidence-bba92a121d5e5d283024",
                "evalevidence-2ec4f8b64f3c3c75e13a"
              ]
            }
          ]
        }
      ]
    },
    {
      "section_id": "claim-ceiling",
      "heading": "What the chain still cannot prove",
      "paragraphs": [
        {
          "sentences": [
            {
              "text": "Hash-chain verification cannot establish that a refusal classifier measures the intended construct, that an endpoint label names fixed hidden weights, or that the panel represents all models.",
              "citation_ids": [
                "evalevidence-2ec4f8b64f3c3c75e13a",
                "evalevidence-80711deb0314c9522bd2"
              ]
            },
            {
              "text": "Those questions require separate validation, provider transparency, and sampling arguments rather than a stronger hash.",
              "citation_ids": [
                "evalevidence-2ec4f8b64f3c3c75e13a",
                "evalevidence-80711deb0314c9522bd2"
              ]
            }
          ]
        }
      ]
    }
  ],
  "counterreadings": [
    {
      "text": "A verified chain is meaningful evidence that the served sequence has not been quietly rewritten.",
      "citation_ids": [
        "evalevidence-2ec4f8b64f3c3c75e13a"
      ]
    },
    {
      "text": "An internally valid chain can still contain a poorly designed eval; integrity and measurement validity are separate gates.",
      "citation_ids": [
        "evalevidence-2ec4f8b64f3c3c75e13a",
        "evalevidence-80711deb0314c9522bd2"
      ]
    }
  ],
  "limitations": [
    {
      "text": "Without an external witness, the registry file alone cannot prove when its first unseen version was published.",
      "citation_ids": [
        "evalevidence-2ec4f8b64f3c3c75e13a"
      ]
    },
    {
      "text": "Registry verification checks commitments and exact metrics but does not independently relabel every model response.",
      "citation_ids": [
        "evalevidence-bba92a121d5e5d283024",
        "evalevidence-80711deb0314c9522bd2"
      ]
    },
    {
      "text": "The endpoint set is a declared panel rather than a probability sample of deployed AI systems.",
      "citation_ids": [
        "evalevidence-80711deb0314c9522bd2"
      ]
    }
  ],
  "methodology": [
    {
      "step": "Freeze",
      "detail": "Append a probe-set commitment before its result can enter the same registry.",
      "citation_ids": [
        "evalevidence-2ec4f8b64f3c3c75e13a"
      ]
    },
    {
      "step": "Link",
      "detail": "Hash every attestation with its predecessor so deletion, reordering, and alteration change the head.",
      "citation_ids": [
        "evalevidence-2ec4f8b64f3c3c75e13a"
      ]
    },
    {
      "step": "Match",
      "detail": "Require every current panel metric to equal its sealed run before building an article.",
      "citation_ids": [
        "evalevidence-bba92a121d5e5d283024",
        "evalevidence-80711deb0314c9522bd2"
      ]
    },
    {
      "step": "Bound",
      "detail": "Keep chain integrity separate from classifier validity, model identity, and population claims.",
      "citation_ids": [
        "evalevidence-2ec4f8b64f3c3c75e13a",
        "evalevidence-80711deb0314c9522bd2"
      ]
    }
  ],
  "evidence": [
    {
      "evidence_id": "evalevidence-2ec4f8b64f3c3c75e13a",
      "input_id": "eval-registry",
      "label": "Verified registry head and complete served-chain summary",
      "selector": "/",
      "value": {
        "attestations": 540,
        "preregistrations": 6,
        "runs": 534,
        "models": [
          "anthropic/claude-3-haiku",
          "deepseek/deepseek-chat",
          "meta-llama/llama-3.1-8b-instruct",
          "meta-llama/llama-3.3-70b-instruct",
          "mistralai/mistral-nemo",
          "openai/gpt-4o-mini",
          "qwen/qwen-2.5-7b-instruct"
        ],
        "head_seq": 539,
        "head_ts": "2026-08-23T13:40:04.986907+00:00",
        "head_hash": "3ee89f1404c1cffb48523fa53fcadc6348038fe93e74aae4633b8b9c81e294c7",
        "merkle_root": "4ffc4893113aaa44a2412fb9d1d82133cd5c79078f30257ddbae240fc89062d5"
      },
      "interpretation_limit": "Verification detects alteration, deletion, or reordering inside the served chain. The file alone does not prove an external wall-clock publication time.",
      "source_url": "https://palimpsest.info/readings/eval-registry.jsonl"
    },
    {
      "evidence_id": "evalevidence-bba92a121d5e5d283024",
      "input_id": "eval-registry",
      "label": "Current panel runs matched to the published reading",
      "selector": "/runs/ts=2026-08-23T13:01:08.138128+00:00",
      "value": [
        {
          "model": "anthropic/claude-3-haiku",
          "seq": 533,
          "suite": "frontier-overrefusal-v2",
          "probe_set_hash": "31208cb5a9f13b3a23bb64d347ed118dc212f39d8b5fe19ac749a7988f7c0adb",
          "responses_hash": "0059ef53893f3c966b7383928c2f4fcecb1cacb01066e5953178cc14a135e671",
          "entry_hash": "e88298440f6c2c1ac41e8b24c78e991f57bf03b134a8c65ef6945470fbf1a80f"
        },
        {
          "model": "meta-llama/llama-3.3-70b-instruct",
          "seq": 535,
          "suite": "frontier-overrefusal-v2",
          "probe_set_hash": "31208cb5a9f13b3a23bb64d347ed118dc212f39d8b5fe19ac749a7988f7c0adb",
          "responses_hash": "4ae4b94ba318e6abdbcef4dccdf4d482e903db8f60bf83c81d603a77783e4f55",
          "entry_hash": "b5ed364a622d770373e5a730c3cfea21d04a56df71509180bfc8de3b7690811b"
        },
        {
          "model": "mistralai/mistral-nemo",
          "seq": 537,
          "suite": "frontier-overrefusal-v2",
          "probe_set_hash": "31208cb5a9f13b3a23bb64d347ed118dc212f39d8b5fe19ac749a7988f7c0adb",
          "responses_hash": "572319ff5caed890c0c5cc8187f159329c53f8b444ba1e8fc096adf89d17a4c4",
          "entry_hash": "685ee001fc38f40351b6b1a5b0d23afecee568d328fb1a0574f1c0a563d7982c"
        },
        {
          "model": "openai/gpt-4o-mini",
          "seq": 539,
          "suite": "frontier-overrefusal-v2",
          "probe_set_hash": "31208cb5a9f13b3a23bb64d347ed118dc212f39d8b5fe19ac749a7988f7c0adb",
          "responses_hash": "5fe18107bbe38377c09b0a820811b9f7ccc7facdc48a6436a4fbdfcd2fe1f497",
          "entry_hash": "3ee89f1404c1cffb48523fa53fcadc6348038fe93e74aae4633b8b9c81e294c7"
        }
      ],
      "interpretation_limit": "Exact metric matching proves publication consistency, not construct validity.",
      "source_url": "https://palimpsest.info/readings/eval-registry.jsonl"
    },
    {
      "evidence_id": "evalevidence-80711deb0314c9522bd2",
      "input_id": "refusal-drift-current",
      "label": "Current suite scope and uncertainty method",
      "selector": "/method",
      "value": {
        "suite": "frontier-overrefusal-v2",
        "arm": "canonical",
        "method": "every question is a family of meaning-preserving wordings; the family is the statistical unit; new refusals (answered->refused) are the erasure events; the standing alarm is a mixture supermartingale, so the lifetime false-alarm rate is bounded under unlimited peeking",
        "method_version": 4,
        "model_count": 4
      },
      "interpretation_limit": "A verified chain does not widen the suite's sampled population.",
      "source_url": "https://palimpsest.info/readings/refusal-drift-latest.json"
    }
  ],
  "evaluation_receipt": {
    "status": "passed",
    "publishable": true,
    "citation_coverage": 1.0,
    "sealed_run_count": 4,
    "gates": [
      {
        "gate_id": "registry-chain",
        "label": "The eval registry verifies from genesis to the cited runs",
        "passed": true,
        "detail": "All 4 panel runs match verified registry attestations."
      },
      {
        "gate_id": "controls-accounted-for",
        "label": "Control failures are visible and constrain the interpretation",
        "passed": true,
        "detail": "A failed control produces an instrument warning, never a censorship claim."
      },
      {
        "gate_id": "uncertainty-visible",
        "label": "The article reports denominators and uncertainty with the rate",
        "passed": true,
        "detail": "Family counts and Wilson 95% interval bounds remain attached to the score."
      },
      {
        "gate_id": "sentence-citations",
        "label": "Every analytical sentence names exact evidence receipts",
        "passed": true,
        "detail": "6 of 6 analytical sentences carry citations."
      },
      {
        "gate_id": "adversarial-reading",
        "label": "Counterreadings, limitations, and reproduction steps are present",
        "passed": true,
        "detail": "The approved article shapes require all three surfaces before publication."
      },
      {
        "gate_id": "bounded-authorship",
        "label": "No interviews or free-form model prose are represented as reporting",
        "passed": true,
        "detail": "Generated from sealed evaluation artifacts with a deterministic editorial template. No interviews and no free-form model prose were used."
      }
    ]
  },
  "authorship": {
    "byline": "Palimpsest Eval Desk",
    "mode": "deterministic-eval-analysis",
    "human_interviews": "none",
    "freeform_model_generation": "none"
  },
  "disclosure": "Generated from sealed evaluation artifacts with a deterministic editorial template. No interviews and no free-form model prose were used."
}
