{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:RNLXS2ABLUR2GE4EKRJ2QCLIGC","short_pith_number":"pith:RNLXS2AB","canonical_record":{"source":{"id":"2502.03461","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-05T18:58:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"4a598477aa05860d6cb0d33ccda774fb1210023a3e8b6090550b4ce8301ca53c","abstract_canon_sha256":"84cd242cad3f6fe6fc1e316e73df4c4af983600be05336c97a1389ce6d8236ea"},"schema_version":"1.0"},"canonical_sha256":"8b577968015d23a313845453a8096830ab9ae4b3edbf82a737c9ca1bd39932d8","source":{"kind":"arxiv","id":"2502.03461","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.03461","created_at":"2026-07-05T10:10:05Z"},{"alias_kind":"arxiv_version","alias_value":"2502.03461v1","created_at":"2026-07-05T10:10:05Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.03461","created_at":"2026-07-05T10:10:05Z"},{"alias_kind":"pith_short_12","alias_value":"RNLXS2ABLUR2","created_at":"2026-07-05T10:10:05Z"},{"alias_kind":"pith_short_16","alias_value":"RNLXS2ABLUR2GE4E","created_at":"2026-07-05T10:10:05Z"},{"alias_kind":"pith_short_8","alias_value":"RNLXS2AB","created_at":"2026-07-05T10:10:05Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:RNLXS2ABLUR2GE4EKRJ2QCLIGC","target":"record","payload":{"canonical_record":{"source":{"id":"2502.03461","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-05T18:58:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"4a598477aa05860d6cb0d33ccda774fb1210023a3e8b6090550b4ce8301ca53c","abstract_canon_sha256":"84cd242cad3f6fe6fc1e316e73df4c4af983600be05336c97a1389ce6d8236ea"},"schema_version":"1.0"},"canonical_sha256":"8b577968015d23a313845453a8096830ab9ae4b3edbf82a737c9ca1bd39932d8","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:10:05.162831Z","signature_b64":"y1dT//TgFGz4gBmY2k2Ik41GZPyjNwOi+fjPdu1tSiwKDkGk5k+7SWbFZAnSaNK30HesoaH7I5jRU5rSA/7uDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b577968015d23a313845453a8096830ab9ae4b3edbf82a737c9ca1bd39932d8","last_reissued_at":"2026-07-05T10:10:05.162366Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:10:05.162366Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2502.03461","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:10:05Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"b3K3Bv+bt8UJmJziL7NcS0aDJKNy5pGnW8enhhbdB5JkPzORiPPTItt2JIi+5FZevo7k90qK5Sny1clLIUStAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T16:52:15.850487Z"},"content_sha256":"5ae0da33afb20da2fb8aa9429c553316ebfedcb7bed729bbc63eeda2e442dc6a","schema_version":"1.0","event_id":"sha256:5ae0da33afb20da2fb8aa9429c553316ebfedcb7bed729bbc63eeda2e442dc6a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:RNLXS2ABLUR2GE4EKRJ2QCLIGC","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Do Large Language Model Benchmarks Test Reliability?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Aleksander Madry, Edward Vendrow, Joshua Vendrow, Sara Beery","submitted_at":"2025-02-05T18:58:19Z","abstract_excerpt":"When deploying large language models (LLMs), it is important to ensure that these models are not only capable, but also reliable. Many benchmarks have been created to track LLMs' growing capabilities, however there has been no similar focus on measuring their reliability. To understand the potential ramifications of this gap, we investigate how well current benchmarks quantify model reliability. We find that pervasive label errors can compromise these evaluations, obscuring lingering model failures and hiding unreliable behavior.\n  Motivated by this gap in the evaluation of reliability, we the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.03461","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.03461/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:10:05Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4Yo+k6nBrwXW1e97c2pjmaeeN31AUKhY3LhINVD3nas0d1mC8qOx/b1JqIYzWsIr6+kdTga+2oI+uVB6txodBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T16:52:15.851357Z"},"content_sha256":"3070aefd6d66bafd8151907f039ba651dbfeb349df85209016315bb2aef5f77c","schema_version":"1.0","event_id":"sha256:3070aefd6d66bafd8151907f039ba651dbfeb349df85209016315bb2aef5f77c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/RNLXS2ABLUR2GE4EKRJ2QCLIGC/bundle.json","state_url":"https://pith.science/pith/RNLXS2ABLUR2GE4EKRJ2QCLIGC/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/RNLXS2ABLUR2GE4EKRJ2QCLIGC/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T16:52:15Z","links":{"resolver":"https://pith.science/pith/RNLXS2ABLUR2GE4EKRJ2QCLIGC","bundle":"https://pith.science/pith/RNLXS2ABLUR2GE4EKRJ2QCLIGC/bundle.json","state":"https://pith.science/pith/RNLXS2ABLUR2GE4EKRJ2QCLIGC/state.json","well_known_bundle":"https://pith.science/.well-known/pith/RNLXS2ABLUR2GE4EKRJ2QCLIGC/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:RNLXS2ABLUR2GE4EKRJ2QCLIGC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"84cd242cad3f6fe6fc1e316e73df4c4af983600be05336c97a1389ce6d8236ea","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-05T18:58:19Z","title_canon_sha256":"4a598477aa05860d6cb0d33ccda774fb1210023a3e8b6090550b4ce8301ca53c"},"schema_version":"1.0","source":{"id":"2502.03461","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.03461","created_at":"2026-07-05T10:10:05Z"},{"alias_kind":"arxiv_version","alias_value":"2502.03461v1","created_at":"2026-07-05T10:10:05Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.03461","created_at":"2026-07-05T10:10:05Z"},{"alias_kind":"pith_short_12","alias_value":"RNLXS2ABLUR2","created_at":"2026-07-05T10:10:05Z"},{"alias_kind":"pith_short_16","alias_value":"RNLXS2ABLUR2GE4E","created_at":"2026-07-05T10:10:05Z"},{"alias_kind":"pith_short_8","alias_value":"RNLXS2AB","created_at":"2026-07-05T10:10:05Z"}],"graph_snapshots":[{"event_id":"sha256:3070aefd6d66bafd8151907f039ba651dbfeb349df85209016315bb2aef5f77c","target":"graph","created_at":"2026-07-05T10:10:05Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.03461/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"When deploying large language models (LLMs), it is important to ensure that these models are not only capable, but also reliable. Many benchmarks have been created to track LLMs' growing capabilities, however there has been no similar focus on measuring their reliability. To understand the potential ramifications of this gap, we investigate how well current benchmarks quantify model reliability. We find that pervasive label errors can compromise these evaluations, obscuring lingering model failures and hiding unreliable behavior.\n  Motivated by this gap in the evaluation of reliability, we the","authors_text":"Aleksander Madry, Edward Vendrow, Joshua Vendrow, Sara Beery","cross_cats":["cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-05T18:58:19Z","title":"Do Large Language Model Benchmarks Test Reliability?"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.03461","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:5ae0da33afb20da2fb8aa9429c553316ebfedcb7bed729bbc63eeda2e442dc6a","target":"record","created_at":"2026-07-05T10:10:05Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"84cd242cad3f6fe6fc1e316e73df4c4af983600be05336c97a1389ce6d8236ea","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-05T18:58:19Z","title_canon_sha256":"4a598477aa05860d6cb0d33ccda774fb1210023a3e8b6090550b4ce8301ca53c"},"schema_version":"1.0","source":{"id":"2502.03461","kind":"arxiv","version":1}},"canonical_sha256":"8b577968015d23a313845453a8096830ab9ae4b3edbf82a737c9ca1bd39932d8","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"8b577968015d23a313845453a8096830ab9ae4b3edbf82a737c9ca1bd39932d8","first_computed_at":"2026-07-05T10:10:05.162366Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:10:05.162366Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"y1dT//TgFGz4gBmY2k2Ik41GZPyjNwOi+fjPdu1tSiwKDkGk5k+7SWbFZAnSaNK30HesoaH7I5jRU5rSA/7uDw==","signature_status":"signed_v1","signed_at":"2026-07-05T10:10:05.162831Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.03461","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:5ae0da33afb20da2fb8aa9429c553316ebfedcb7bed729bbc63eeda2e442dc6a","sha256:3070aefd6d66bafd8151907f039ba651dbfeb349df85209016315bb2aef5f77c"],"state_sha256":"bd0b7c60438d9707ecc345c354aae672ffbd5a5e0ce19487ab4f00a9de8bf760"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"XgY7JQ1KNm/fogjLzwUvw5fK5d7XCK17tcoyl9/nHfuXQk7uFGJi72CVeuiUURGs3jrctee9NQ37jCaeASmBCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T16:52:15.855648Z","bundle_sha256":"dbf9203be5fde47a867ab3b665aa3cc0a2de503689b0ba7a81a45a621f33c789"}}