{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:VNNHU5KKX3Q3PRAAGPB6OVZPJI","short_pith_number":"pith:VNNHU5KK","canonical_record":{"source":{"id":"2406.08446","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-12T17:37:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a3073db236b32714846d4c038ac234a90ed27952e75ef47ebadba14119f55556","abstract_canon_sha256":"64060251573726c1f1d8f41e972fe287c46d5968d00bd3cfd78e311ae1c43f69"},"schema_version":"1.0"},"canonical_sha256":"ab5a7a754abee1b7c40033c3e7572f4a2dba08bf2d9bee0d083200fa19f7439c","source":{"kind":"arxiv","id":"2406.08446","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.08446","created_at":"2026-07-05T10:12:48Z"},{"alias_kind":"arxiv_version","alias_value":"2406.08446v2","created_at":"2026-07-05T10:12:48Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08446","created_at":"2026-07-05T10:12:48Z"},{"alias_kind":"pith_short_12","alias_value":"VNNHU5KKX3Q3","created_at":"2026-07-05T10:12:48Z"},{"alias_kind":"pith_short_16","alias_value":"VNNHU5KKX3Q3PRAA","created_at":"2026-07-05T10:12:48Z"},{"alias_kind":"pith_short_8","alias_value":"VNNHU5KK","created_at":"2026-07-05T10:12:48Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:VNNHU5KKX3Q3PRAAGPB6OVZPJI","target":"record","payload":{"canonical_record":{"source":{"id":"2406.08446","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-12T17:37:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a3073db236b32714846d4c038ac234a90ed27952e75ef47ebadba14119f55556","abstract_canon_sha256":"64060251573726c1f1d8f41e972fe287c46d5968d00bd3cfd78e311ae1c43f69"},"schema_version":"1.0"},"canonical_sha256":"ab5a7a754abee1b7c40033c3e7572f4a2dba08bf2d9bee0d083200fa19f7439c","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:48.148278Z","signature_b64":"0l9Ziu13Xnb3NZ5A2NQFGadf03G8s91q+PD4B7pGX642uhrUofO8XB5klrO+SvnB8D8/uBJ/2BXx5qK1D+zZBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ab5a7a754abee1b7c40033c3e7572f4a2dba08bf2d9bee0d083200fa19f7439c","last_reissued_at":"2026-07-05T10:12:48.147779Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:48.147779Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2406.08446","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:12:48Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"t7fQzwgU8b5GFwfiU3AYhgmDQtaU6Lvu/eE8cvk4p70CaTPZ4haQqKEs1IAmuC6yn1Wi43fFRaQ1W7XPnGP0BA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T02:41:07.757240Z"},"content_sha256":"a10de6915ad517143596b8b817d6560f8d21543bb2159dd80ae14776a201dab6","schema_version":"1.0","event_id":"sha256:a10de6915ad517143596b8b817d6560f8d21543bb2159dd80ae14776a201dab6"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:VNNHU5KKX3Q3PRAAGPB6OVZPJI","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"OLMES: A Standard for Language Model Evaluations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bailey Kuehl, Dany Haddad, Hannaneh Hajishirzi, Jesse Dodge, Oyvind Tafjord, Yuling Gu","submitted_at":"2024-06-12T17:37:09Z","abstract_excerpt":"Progress in AI is often demonstrated by new models claiming improved performance on tasks measuring model capabilities. Evaluating language models can be particularly challenging, as choices of how a model is evaluated on a task can lead to large changes in measured performance. There is no common standard setup, so different models are evaluated on the same tasks in different ways, leading to claims about which models perform best not being reproducible. We propose OLMES, a completely documented, practical, open standard for reproducible LLM evaluations. In developing this standard, we identi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08446","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08446/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:12:48Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"k4b+kH6P2TiMbr2CU71JF2szbjLwlDoHREAJE6/BYpW3JUWZGAsfzUlST6+td6NrE11C/V9+fVGjyFD16KWcCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T02:41:07.757748Z"},"content_sha256":"51b0127f1a90dfc0b651a991feaeddba015eeeb26fe884a972e6b9ac8e78e979","schema_version":"1.0","event_id":"sha256:51b0127f1a90dfc0b651a991feaeddba015eeeb26fe884a972e6b9ac8e78e979"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI/bundle.json","state_url":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T02:41:07Z","links":{"resolver":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI","bundle":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI/bundle.json","state":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI/state.json","well_known_bundle":"https://pith.science/.well-known/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:VNNHU5KKX3Q3PRAAGPB6OVZPJI","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"64060251573726c1f1d8f41e972fe287c46d5968d00bd3cfd78e311ae1c43f69","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-12T17:37:09Z","title_canon_sha256":"a3073db236b32714846d4c038ac234a90ed27952e75ef47ebadba14119f55556"},"schema_version":"1.0","source":{"id":"2406.08446","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.08446","created_at":"2026-07-05T10:12:48Z"},{"alias_kind":"arxiv_version","alias_value":"2406.08446v2","created_at":"2026-07-05T10:12:48Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08446","created_at":"2026-07-05T10:12:48Z"},{"alias_kind":"pith_short_12","alias_value":"VNNHU5KKX3Q3","created_at":"2026-07-05T10:12:48Z"},{"alias_kind":"pith_short_16","alias_value":"VNNHU5KKX3Q3PRAA","created_at":"2026-07-05T10:12:48Z"},{"alias_kind":"pith_short_8","alias_value":"VNNHU5KK","created_at":"2026-07-05T10:12:48Z"}],"graph_snapshots":[{"event_id":"sha256:51b0127f1a90dfc0b651a991feaeddba015eeeb26fe884a972e6b9ac8e78e979","target":"graph","created_at":"2026-07-05T10:12:48Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2406.08446/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Progress in AI is often demonstrated by new models claiming improved performance on tasks measuring model capabilities. Evaluating language models can be particularly challenging, as choices of how a model is evaluated on a task can lead to large changes in measured performance. There is no common standard setup, so different models are evaluated on the same tasks in different ways, leading to claims about which models perform best not being reproducible. We propose OLMES, a completely documented, practical, open standard for reproducible LLM evaluations. In developing this standard, we identi","authors_text":"Bailey Kuehl, Dany Haddad, Hannaneh Hajishirzi, Jesse Dodge, Oyvind Tafjord, Yuling Gu","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-12T17:37:09Z","title":"OLMES: A Standard for Language Model Evaluations"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08446","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:a10de6915ad517143596b8b817d6560f8d21543bb2159dd80ae14776a201dab6","target":"record","created_at":"2026-07-05T10:12:48Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"64060251573726c1f1d8f41e972fe287c46d5968d00bd3cfd78e311ae1c43f69","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-12T17:37:09Z","title_canon_sha256":"a3073db236b32714846d4c038ac234a90ed27952e75ef47ebadba14119f55556"},"schema_version":"1.0","source":{"id":"2406.08446","kind":"arxiv","version":2}},"canonical_sha256":"ab5a7a754abee1b7c40033c3e7572f4a2dba08bf2d9bee0d083200fa19f7439c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ab5a7a754abee1b7c40033c3e7572f4a2dba08bf2d9bee0d083200fa19f7439c","first_computed_at":"2026-07-05T10:12:48.147779Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:12:48.147779Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"0l9Ziu13Xnb3NZ5A2NQFGadf03G8s91q+PD4B7pGX642uhrUofO8XB5klrO+SvnB8D8/uBJ/2BXx5qK1D+zZBg==","signature_status":"signed_v1","signed_at":"2026-07-05T10:12:48.148278Z","signed_message":"canonical_sha256_bytes"},"source_id":"2406.08446","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:a10de6915ad517143596b8b817d6560f8d21543bb2159dd80ae14776a201dab6","sha256:51b0127f1a90dfc0b651a991feaeddba015eeeb26fe884a972e6b9ac8e78e979"],"state_sha256":"45df17ebd2e9ab9a828480a3dc7207abdf11af0f60947ef293537e4eabbf2bff"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"q+ZzvXgRGFKCnJXWl1Q1ARphamtRzqj5aL0KVjpNK5Bog1hyu1fcoghtTcPkaEY5pGk9Ms6+z3u6DNKjFejPBQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T02:41:07.765668Z","bundle_sha256":"7133a306ec443de155446a64f22689e2fdc6f6a680fef24a72083c12cba5b52e"}}