{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:CRNS7LGMCSHJZ6IM7BF5WCESGL","short_pith_number":"pith:CRNS7LGM","canonical_record":{"source":{"id":"2605.06213","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-05-07T13:15:31Z","cross_cats_sorted":[],"title_canon_sha256":"a710aa693ed0ba3acae331e88927f7353b96133d743204884b058c359ec47750","abstract_canon_sha256":"8fc4fa04fdb801a1bec4f8acff3dfb7c2bcd45a5d807a3ca47cfc98e1e544b09"},"schema_version":"1.0"},"canonical_sha256":"145b2faccc148e9cf90cf84bdb089232dda0f6a930a35d3aceefc0edcc31e564","source":{"kind":"arxiv","id":"2605.06213","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.06213","created_at":"2026-05-27T02:05:21Z"},{"alias_kind":"arxiv_version","alias_value":"2605.06213v2","created_at":"2026-05-27T02:05:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.06213","created_at":"2026-05-27T02:05:21Z"},{"alias_kind":"pith_short_12","alias_value":"CRNS7LGMCSHJ","created_at":"2026-05-27T02:05:21Z"},{"alias_kind":"pith_short_16","alias_value":"CRNS7LGMCSHJZ6IM","created_at":"2026-05-27T02:05:21Z"},{"alias_kind":"pith_short_8","alias_value":"CRNS7LGM","created_at":"2026-05-27T02:05:21Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:CRNS7LGMCSHJZ6IM7BF5WCESGL","target":"record","payload":{"canonical_record":{"source":{"id":"2605.06213","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-05-07T13:15:31Z","cross_cats_sorted":[],"title_canon_sha256":"a710aa693ed0ba3acae331e88927f7353b96133d743204884b058c359ec47750","abstract_canon_sha256":"8fc4fa04fdb801a1bec4f8acff3dfb7c2bcd45a5d807a3ca47cfc98e1e544b09"},"schema_version":"1.0"},"canonical_sha256":"145b2faccc148e9cf90cf84bdb089232dda0f6a930a35d3aceefc0edcc31e564","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-27T02:05:21.281310Z","signature_b64":"WP7ytSwUn/rKajUEjHblXR2ZBZTltuz3JW8S/62u8d/Vlgit19G6yZVa4fq3QGuXr7wJp3FuKh2BbycksdJvCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"145b2faccc148e9cf90cf84bdb089232dda0f6a930a35d3aceefc0edcc31e564","last_reissued_at":"2026-05-27T02:05:21.280792Z","signature_status":"signed_v1","first_computed_at":"2026-05-27T02:05:21.280792Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2605.06213","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-27T02:05:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IeZlDFODSUpszH+9lazd4v1XmPGtOU7xPxOgnQ2b3IvLI73czRkYgOyGC3rAgNo6LvoCBNAT7aV9TqbIVdoABw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T03:49:19.549497Z"},"content_sha256":"4d47f88bab55f0fce7b769474c98201bbe6d710638c4eb7cfbabb1ce108b0998","schema_version":"1.0","event_id":"sha256:4d47f88bab55f0fce7b769474c98201bbe6d710638c4eb7cfbabb1ce108b0998"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:CRNS7LGMCSHJZ6IM7BF5WCESGL","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Beyond Fixed Benchmarks and Worst-Case Attacks: Dynamic Boundary Evaluation for Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Evaluating language models at each model's 0.5 success probability boundary reveals capability gaps that fixed benchmarks miss.","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Da Yu, Haoxiang Wang, Huishuai Zhang","submitted_at":"2026-05-07T13:15:31Z","abstract_excerpt":"Evaluating large language models (LLMs) today rests on fixed benchmarks that apply the same set of items to any model, producing ceiling and floor effects that mask capability gaps. We argue that the most informative evaluation signal lies at the boundary, where the per-prompt pass probability is near $0.5$ under random-sampling decoding, and propose Dynamic Boundary Evaluation (DBE), which actively locates each model's boundary and places it on a globally comparable difficulty scale. DBE delivers three artifacts: (i) a calibrated item bank covering safety, capability, and truthfulness, with p"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"We argue that the most informative evaluation signal lies at the boundary, where the per-prompt pass probability is near 0.5 under random-sampling decoding, and propose Dynamic Boundary Evaluation (DBE), which actively locates each model's boundary and places it on a globally comparable difficulty scale.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the per-item difficulty labels validated across 9 reference LLMs will allow accurate placement of new models on the same scale, and that the boundary at 0.5 probability is indeed the most informative point for evaluation.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Dynamic Boundary Evaluation adaptively identifies each LLM's performance boundary on a shared difficulty scale using a calibrated item bank and a search algorithm.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Evaluating language models at each model's 0.5 success probability boundary reveals capability gaps that fixed benchmarks miss.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"030822e8eeb150e0adf494a40577f43f2faf8c818942aac775bfc1edc3638da2"},"source":{"id":"2605.06213","kind":"arxiv","version":2},"verdict":{"id":"7810e70b-df20-4e0e-af64-f16d0a2bdf43","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-08T10:08:34.907950Z","strongest_claim":"We argue that the most informative evaluation signal lies at the boundary, where the per-prompt pass probability is near 0.5 under random-sampling decoding, and propose Dynamic Boundary Evaluation (DBE), which actively locates each model's boundary and places it on a globally comparable difficulty scale.","one_line_summary":"Dynamic Boundary Evaluation adaptively identifies each LLM's performance boundary on a shared difficulty scale using a calibrated item bank and a search algorithm.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the per-item difficulty labels validated across 9 reference LLMs will allow accurate placement of new models on the same scale, and that the boundary at 0.5 probability is indeed the most informative point for evaluation.","pith_extraction_headline":"Evaluating language models at each model's 0.5 success probability boundary reveals capability gaps that fixed benchmarks miss."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2605.06213/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"claim_evidence","ran_at":"2026-05-20T13:02:04.225829Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"ai_meta_artifact","ran_at":"2026-05-20T08:35:04.914058Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_title_agreement","ran_at":"2026-05-19T19:01:19.144676Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T12:53:40.068746Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"105fc5b0074d3b4c0612e243b40dc4ed6b5d35f0346d873c1f0d654f56b2f056"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"7810e70b-df20-4e0e-af64-f16d0a2bdf43"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-27T02:05:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ja9dlYrV6LXgHnPr1qZR961dUhiu/OfMZlZlZlRuS6OnLUZ46wII5Pc0Bd6q9DJWr3qoID7TVpkE1esDt4KPAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T03:49:19.550207Z"},"content_sha256":"9015ea7acf064276f0a46231922272960e5a9b458a7903fc1df8ce3453781b17","schema_version":"1.0","event_id":"sha256:9015ea7acf064276f0a46231922272960e5a9b458a7903fc1df8ce3453781b17"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/CRNS7LGMCSHJZ6IM7BF5WCESGL/bundle.json","state_url":"https://pith.science/pith/CRNS7LGMCSHJZ6IM7BF5WCESGL/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/CRNS7LGMCSHJZ6IM7BF5WCESGL/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T03:49:19Z","links":{"resolver":"https://pith.science/pith/CRNS7LGMCSHJZ6IM7BF5WCESGL","bundle":"https://pith.science/pith/CRNS7LGMCSHJZ6IM7BF5WCESGL/bundle.json","state":"https://pith.science/pith/CRNS7LGMCSHJZ6IM7BF5WCESGL/state.json","well_known_bundle":"https://pith.science/.well-known/pith/CRNS7LGMCSHJZ6IM7BF5WCESGL/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:CRNS7LGMCSHJZ6IM7BF5WCESGL","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8fc4fa04fdb801a1bec4f8acff3dfb7c2bcd45a5d807a3ca47cfc98e1e544b09","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-05-07T13:15:31Z","title_canon_sha256":"a710aa693ed0ba3acae331e88927f7353b96133d743204884b058c359ec47750"},"schema_version":"1.0","source":{"id":"2605.06213","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.06213","created_at":"2026-05-27T02:05:21Z"},{"alias_kind":"arxiv_version","alias_value":"2605.06213v2","created_at":"2026-05-27T02:05:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.06213","created_at":"2026-05-27T02:05:21Z"},{"alias_kind":"pith_short_12","alias_value":"CRNS7LGMCSHJ","created_at":"2026-05-27T02:05:21Z"},{"alias_kind":"pith_short_16","alias_value":"CRNS7LGMCSHJZ6IM","created_at":"2026-05-27T02:05:21Z"},{"alias_kind":"pith_short_8","alias_value":"CRNS7LGM","created_at":"2026-05-27T02:05:21Z"}],"graph_snapshots":[{"event_id":"sha256:9015ea7acf064276f0a46231922272960e5a9b458a7903fc1df8ce3453781b17","target":"graph","created_at":"2026-05-27T02:05:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"We argue that the most informative evaluation signal lies at the boundary, where the per-prompt pass probability is near 0.5 under random-sampling decoding, and propose Dynamic Boundary Evaluation (DBE), which actively locates each model's boundary and places it on a globally comparable difficulty scale."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the per-item difficulty labels validated across 9 reference LLMs will allow accurate placement of new models on the same scale, and that the boundary at 0.5 probability is indeed the most informative point for evaluation."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Dynamic Boundary Evaluation adaptively identifies each LLM's performance boundary on a shared difficulty scale using a calibrated item bank and a search algorithm."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Evaluating language models at each model's 0.5 success probability boundary reveals capability gaps that fixed benchmarks miss."}],"snapshot_sha256":"030822e8eeb150e0adf494a40577f43f2faf8c818942aac775bfc1edc3638da2"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"claim_evidence","ran_at":"2026-05-20T13:02:04.225829Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-20T08:35:04.914058Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_title_agreement","ran_at":"2026-05-19T19:01:19.144676Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-19T12:53:40.068746Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2605.06213/integrity.json","findings":[],"snapshot_sha256":"105fc5b0074d3b4c0612e243b40dc4ed6b5d35f0346d873c1f0d654f56b2f056","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Evaluating large language models (LLMs) today rests on fixed benchmarks that apply the same set of items to any model, producing ceiling and floor effects that mask capability gaps. We argue that the most informative evaluation signal lies at the boundary, where the per-prompt pass probability is near $0.5$ under random-sampling decoding, and propose Dynamic Boundary Evaluation (DBE), which actively locates each model's boundary and places it on a globally comparable difficulty scale. DBE delivers three artifacts: (i) a calibrated item bank covering safety, capability, and truthfulness, with p","authors_text":"Da Yu, Haoxiang Wang, Huishuai Zhang","cross_cats":[],"headline":"Evaluating language models at each model's 0.5 success probability boundary reveals capability gaps that fixed benchmarks miss.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-05-07T13:15:31Z","title":"Beyond Fixed Benchmarks and Worst-Case Attacks: Dynamic Boundary Evaluation for Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2605.06213","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-08T10:08:34.907950Z","id":"7810e70b-df20-4e0e-af64-f16d0a2bdf43","model_set":{"reader":"grok-4.3"},"one_line_summary":"Dynamic Boundary Evaluation adaptively identifies each LLM's performance boundary on a shared difficulty scale using a calibrated item bank and a search algorithm.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Evaluating language models at each model's 0.5 success probability boundary reveals capability gaps that fixed benchmarks miss.","strongest_claim":"We argue that the most informative evaluation signal lies at the boundary, where the per-prompt pass probability is near 0.5 under random-sampling decoding, and propose Dynamic Boundary Evaluation (DBE), which actively locates each model's boundary and places it on a globally comparable difficulty scale.","weakest_assumption":"That the per-item difficulty labels validated across 9 reference LLMs will allow accurate placement of new models on the same scale, and that the boundary at 0.5 probability is indeed the most informative point for evaluation."}},"verdict_id":"7810e70b-df20-4e0e-af64-f16d0a2bdf43"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4d47f88bab55f0fce7b769474c98201bbe6d710638c4eb7cfbabb1ce108b0998","target":"record","created_at":"2026-05-27T02:05:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8fc4fa04fdb801a1bec4f8acff3dfb7c2bcd45a5d807a3ca47cfc98e1e544b09","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-05-07T13:15:31Z","title_canon_sha256":"a710aa693ed0ba3acae331e88927f7353b96133d743204884b058c359ec47750"},"schema_version":"1.0","source":{"id":"2605.06213","kind":"arxiv","version":2}},"canonical_sha256":"145b2faccc148e9cf90cf84bdb089232dda0f6a930a35d3aceefc0edcc31e564","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"145b2faccc148e9cf90cf84bdb089232dda0f6a930a35d3aceefc0edcc31e564","first_computed_at":"2026-05-27T02:05:21.280792Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-27T02:05:21.280792Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"WP7ytSwUn/rKajUEjHblXR2ZBZTltuz3JW8S/62u8d/Vlgit19G6yZVa4fq3QGuXr7wJp3FuKh2BbycksdJvCg==","signature_status":"signed_v1","signed_at":"2026-05-27T02:05:21.281310Z","signed_message":"canonical_sha256_bytes"},"source_id":"2605.06213","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4d47f88bab55f0fce7b769474c98201bbe6d710638c4eb7cfbabb1ce108b0998","sha256:9015ea7acf064276f0a46231922272960e5a9b458a7903fc1df8ce3453781b17"],"state_sha256":"1d577e25cc2a25600c946d09c8f60ea91f3f1bf45c50cd75f4da1ba666d7b4a1"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"yCTrAUXyuPRHF6uyAFCs4uzEPQonDx3ssJNPwhUb4dOW2xIiCNZnPSpBXGQvkaWWL/DriiFsMpm1N4/04yhoAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T03:49:19.554474Z","bundle_sha256":"a64dbac4b5e15a63d84ba1f4721c754f803574a4d97bb50ccd59fdb3328be93c"}}