{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:OQ27JHWAS2D2E3HK5PRG3MP75I","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e97c53698322eb391cc3bdfd42c5a526b5afeeeb88c7d3d887663dac94d5adbe","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-16T19:09:02Z","title_canon_sha256":"bfac478c9eb5a577b6613164fc8e5a268c8e66aaacfcefc717769df9a25a6c47"},"schema_version":"1.0","source":{"id":"2410.12974","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.12974","created_at":"2026-07-05T11:14:25Z"},{"alias_kind":"arxiv_version","alias_value":"2410.12974v3","created_at":"2026-07-05T11:14:25Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12974","created_at":"2026-07-05T11:14:25Z"},{"alias_kind":"pith_short_12","alias_value":"OQ27JHWAS2D2","created_at":"2026-07-05T11:14:25Z"},{"alias_kind":"pith_short_16","alias_value":"OQ27JHWAS2D2E3HK","created_at":"2026-07-05T11:14:25Z"},{"alias_kind":"pith_short_8","alias_value":"OQ27JHWA","created_at":"2026-07-05T11:14:25Z"}],"graph_snapshots":[{"event_id":"sha256:41399a7961961c66ce5566e5619e20484ccf3bf6c1e24e2eff9ba4fb2aafb2cd","target":"graph","created_at":"2026-07-05T11:14:25Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2410.12974/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large language models (LLMs) are powerful tools capable of handling diverse tasks. Comparing and selecting appropriate LLMs for specific tasks requires systematic evaluation methods, as models exhibit varying capabilities across different domains. However, finding suitable benchmarks is difficult given the many available options. This complexity not only increases the risk of benchmark misuse and misinterpretation but also demands substantial effort from LLM users, seeking the most suitable benchmarks for their specific needs. To address these issues, we introduce \\texttt{BenchmarkCards}, an i","authors_text":"Anna Sokol, David Piorkowski, Elizabeth Daly, Michael Hind, Nitesh Chawla, Nuno Moniz, Xiangliang Zhang","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-16T19:09:02Z","title":"BenchmarkCards: Standardized Documentation for Large Language Model Benchmarks"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12974","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3009a59e5bb3b23335419e1d0383b81a807317fd07f11de3cb290a522a11db40","target":"record","created_at":"2026-07-05T11:14:25Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e97c53698322eb391cc3bdfd42c5a526b5afeeeb88c7d3d887663dac94d5adbe","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-16T19:09:02Z","title_canon_sha256":"bfac478c9eb5a577b6613164fc8e5a268c8e66aaacfcefc717769df9a25a6c47"},"schema_version":"1.0","source":{"id":"2410.12974","kind":"arxiv","version":3}},"canonical_sha256":"7435f49ec09687a26ceaebe26db1ffea1ce256c2bcedfef980182524d7622319","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"7435f49ec09687a26ceaebe26db1ffea1ce256c2bcedfef980182524d7622319","first_computed_at":"2026-07-05T11:14:25.230121Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:14:25.230121Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"WT7DEABwrPTvF+30odnEmR2X0FESOt4eVqDZsZH9KnpTZKCEt8BsU84evrYeb4Nrh9G/r1SnZq2tEfKMGfGoDg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:14:25.230626Z","signed_message":"canonical_sha256_bytes"},"source_id":"2410.12974","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3009a59e5bb3b23335419e1d0383b81a807317fd07f11de3cb290a522a11db40","sha256:41399a7961961c66ce5566e5619e20484ccf3bf6c1e24e2eff9ba4fb2aafb2cd"],"state_sha256":"5f65bbc9aab4ec22af3930cbe1e79ce700c828be3030292ebd1164d67aedd941"}