{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OQ27JHWAS2D2E3HK5PRG3MP75I","short_pith_number":"pith:OQ27JHWA","schema_version":"1.0","canonical_sha256":"7435f49ec09687a26ceaebe26db1ffea1ce256c2bcedfef980182524d7622319","source":{"kind":"arxiv","id":"2410.12974","version":3},"attestation_state":"computed","paper":{"title":"BenchmarkCards: Standardized Documentation for Large Language Model Benchmarks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anna Sokol, David Piorkowski, Elizabeth Daly, Michael Hind, Nitesh Chawla, Nuno Moniz, Xiangliang Zhang","submitted_at":"2024-10-16T19:09:02Z","abstract_excerpt":"Large language models (LLMs) are powerful tools capable of handling diverse tasks. Comparing and selecting appropriate LLMs for specific tasks requires systematic evaluation methods, as models exhibit varying capabilities across different domains. However, finding suitable benchmarks is difficult given the many available options. This complexity not only increases the risk of benchmark misuse and misinterpretation but also demands substantial effort from LLM users, seeking the most suitable benchmarks for their specific needs. To address these issues, we introduce \\texttt{BenchmarkCards}, an i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.12974","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-16T19:09:02Z","cross_cats_sorted":[],"title_canon_sha256":"bfac478c9eb5a577b6613164fc8e5a268c8e66aaacfcefc717769df9a25a6c47","abstract_canon_sha256":"e97c53698322eb391cc3bdfd42c5a526b5afeeeb88c7d3d887663dac94d5adbe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:25.230626Z","signature_b64":"WT7DEABwrPTvF+30odnEmR2X0FESOt4eVqDZsZH9KnpTZKCEt8BsU84evrYeb4Nrh9G/r1SnZq2tEfKMGfGoDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7435f49ec09687a26ceaebe26db1ffea1ce256c2bcedfef980182524d7622319","last_reissued_at":"2026-07-05T11:14:25.230121Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:25.230121Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BenchmarkCards: Standardized Documentation for Large Language Model Benchmarks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anna Sokol, David Piorkowski, Elizabeth Daly, Michael Hind, Nitesh Chawla, Nuno Moniz, Xiangliang Zhang","submitted_at":"2024-10-16T19:09:02Z","abstract_excerpt":"Large language models (LLMs) are powerful tools capable of handling diverse tasks. Comparing and selecting appropriate LLMs for specific tasks requires systematic evaluation methods, as models exhibit varying capabilities across different domains. However, finding suitable benchmarks is difficult given the many available options. This complexity not only increases the risk of benchmark misuse and misinterpretation but also demands substantial effort from LLM users, seeking the most suitable benchmarks for their specific needs. To address these issues, we introduce \\texttt{BenchmarkCards}, an i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12974","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.12974/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.12974","created_at":"2026-07-05T11:14:25.230177+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.12974v3","created_at":"2026-07-05T11:14:25.230177+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12974","created_at":"2026-07-05T11:14:25.230177+00:00"},{"alias_kind":"pith_short_12","alias_value":"OQ27JHWAS2D2","created_at":"2026-07-05T11:14:25.230177+00:00"},{"alias_kind":"pith_short_16","alias_value":"OQ27JHWAS2D2E3HK","created_at":"2026-07-05T11:14:25.230177+00:00"},{"alias_kind":"pith_short_8","alias_value":"OQ27JHWA","created_at":"2026-07-05T11:14:25.230177+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.14516","citing_title":"Every Eval Ever: A Unifying Schema and Community Repository for AI Evaluation Results","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12207","citing_title":"Intelligent Automation for Embodied Benchmark Construction: Pipelines, Embodiments, Simulators, and Trends","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09617","citing_title":"AdaQE-CG: Adaptive Query Expansion for Web-Scale Generative AI Model and Data Card Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24902","citing_title":"Safety Drift After Fine-Tuning: Evidence from High-Stakes Domains","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OQ27JHWAS2D2E3HK5PRG3MP75I","json":"https://pith.science/pith/OQ27JHWAS2D2E3HK5PRG3MP75I.json","graph_json":"https://pith.science/api/pith-number/OQ27JHWAS2D2E3HK5PRG3MP75I/graph.json","events_json":"https://pith.science/api/pith-number/OQ27JHWAS2D2E3HK5PRG3MP75I/events.json","paper":"https://pith.science/paper/OQ27JHWA"},"agent_actions":{"view_html":"https://pith.science/pith/OQ27JHWAS2D2E3HK5PRG3MP75I","download_json":"https://pith.science/pith/OQ27JHWAS2D2E3HK5PRG3MP75I.json","view_paper":"https://pith.science/paper/OQ27JHWA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.12974&json=true","fetch_graph":"https://pith.science/api/pith-number/OQ27JHWAS2D2E3HK5PRG3MP75I/graph.json","fetch_events":"https://pith.science/api/pith-number/OQ27JHWAS2D2E3HK5PRG3MP75I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OQ27JHWAS2D2E3HK5PRG3MP75I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OQ27JHWAS2D2E3HK5PRG3MP75I/action/storage_attestation","attest_author":"https://pith.science/pith/OQ27JHWAS2D2E3HK5PRG3MP75I/action/author_attestation","sign_citation":"https://pith.science/pith/OQ27JHWAS2D2E3HK5PRG3MP75I/action/citation_signature","submit_replication":"https://pith.science/pith/OQ27JHWAS2D2E3HK5PRG3MP75I/action/replication_record"}},"created_at":"2026-07-05T11:14:25.230177+00:00","updated_at":"2026-07-05T11:14:25.230177+00:00"}