{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VNNHU5KKX3Q3PRAAGPB6OVZPJI","short_pith_number":"pith:VNNHU5KK","schema_version":"1.0","canonical_sha256":"ab5a7a754abee1b7c40033c3e7572f4a2dba08bf2d9bee0d083200fa19f7439c","source":{"kind":"arxiv","id":"2406.08446","version":2},"attestation_state":"computed","paper":{"title":"OLMES: A Standard for Language Model Evaluations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bailey Kuehl, Dany Haddad, Hannaneh Hajishirzi, Jesse Dodge, Oyvind Tafjord, Yuling Gu","submitted_at":"2024-06-12T17:37:09Z","abstract_excerpt":"Progress in AI is often demonstrated by new models claiming improved performance on tasks measuring model capabilities. Evaluating language models can be particularly challenging, as choices of how a model is evaluated on a task can lead to large changes in measured performance. There is no common standard setup, so different models are evaluated on the same tasks in different ways, leading to claims about which models perform best not being reproducible. We propose OLMES, a completely documented, practical, open standard for reproducible LLM evaluations. In developing this standard, we identi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08446","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-12T17:37:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a3073db236b32714846d4c038ac234a90ed27952e75ef47ebadba14119f55556","abstract_canon_sha256":"64060251573726c1f1d8f41e972fe287c46d5968d00bd3cfd78e311ae1c43f69"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:48.148278Z","signature_b64":"0l9Ziu13Xnb3NZ5A2NQFGadf03G8s91q+PD4B7pGX642uhrUofO8XB5klrO+SvnB8D8/uBJ/2BXx5qK1D+zZBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ab5a7a754abee1b7c40033c3e7572f4a2dba08bf2d9bee0d083200fa19f7439c","last_reissued_at":"2026-07-05T10:12:48.147779Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:48.147779Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OLMES: A Standard for Language Model Evaluations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bailey Kuehl, Dany Haddad, Hannaneh Hajishirzi, Jesse Dodge, Oyvind Tafjord, Yuling Gu","submitted_at":"2024-06-12T17:37:09Z","abstract_excerpt":"Progress in AI is often demonstrated by new models claiming improved performance on tasks measuring model capabilities. Evaluating language models can be particularly challenging, as choices of how a model is evaluated on a task can lead to large changes in measured performance. There is no common standard setup, so different models are evaluated on the same tasks in different ways, leading to claims about which models perform best not being reproducible. We propose OLMES, a completely documented, practical, open standard for reproducible LLM evaluations. In developing this standard, we identi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08446","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08446/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08446","created_at":"2026-07-05T10:12:48.147837+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08446v2","created_at":"2026-07-05T10:12:48.147837+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08446","created_at":"2026-07-05T10:12:48.147837+00:00"},{"alias_kind":"pith_short_12","alias_value":"VNNHU5KKX3Q3","created_at":"2026-07-05T10:12:48.147837+00:00"},{"alias_kind":"pith_short_16","alias_value":"VNNHU5KKX3Q3PRAA","created_at":"2026-07-05T10:12:48.147837+00:00"},{"alias_kind":"pith_short_8","alias_value":"VNNHU5KK","created_at":"2026-07-05T10:12:48.147837+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02266","citing_title":"HERMES: A Multi-Granularity Labeling Substrate for Pre-training Data Mixtures","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09165","citing_title":"Sparse Layers are Critical to Scaling Looped Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02737","citing_title":"SmolLM2: When Smol Goes Big -- Data-Centric Training of a Small Language Model","ref_index":175,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09630","citing_title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09165","citing_title":"Sparse Layers are Critical to Scaling Looped Language Models","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI","json":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI.json","graph_json":"https://pith.science/api/pith-number/VNNHU5KKX3Q3PRAAGPB6OVZPJI/graph.json","events_json":"https://pith.science/api/pith-number/VNNHU5KKX3Q3PRAAGPB6OVZPJI/events.json","paper":"https://pith.science/paper/VNNHU5KK"},"agent_actions":{"view_html":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI","download_json":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI.json","view_paper":"https://pith.science/paper/VNNHU5KK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08446&json=true","fetch_graph":"https://pith.science/api/pith-number/VNNHU5KKX3Q3PRAAGPB6OVZPJI/graph.json","fetch_events":"https://pith.science/api/pith-number/VNNHU5KKX3Q3PRAAGPB6OVZPJI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI/action/storage_attestation","attest_author":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI/action/author_attestation","sign_citation":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI/action/citation_signature","submit_replication":"https://pith.science/pith/VNNHU5KKX3Q3PRAAGPB6OVZPJI/action/replication_record"}},"created_at":"2026-07-05T10:12:48.147837+00:00","updated_at":"2026-07-05T10:12:48.147837+00:00"}