{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YLWS5IMSFEZWWHZBSE7I3J53KD","short_pith_number":"pith:YLWS5IMS","schema_version":"1.0","canonical_sha256":"c2ed2ea19229336b1f21913e8da7bb50f0887d4fc43e53f1bfbf5a4a591f43a4","source":{"kind":"arxiv","id":"2307.09701","version":1},"attestation_state":"computed","paper":{"title":"Efficiency Pentathlon: A Standardized Arena for Efficiency Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Darrell Plessas, Emma Strubell, Evan Pete Walsh, Hannaneh Hajishirzi, Hao Peng, Iz Beltagy, Jared Fernandez, Jesse Dodge, Kyle Lo, Matthew E. Peters, Noah A. Smith, Qingqing Cao, Sam Skjonsberg, Tom Sherborne","submitted_at":"2023-07-19T01:05:33Z","abstract_excerpt":"Rising computational demands of modern natural language processing (NLP) systems have increased the barrier to entry for cutting-edge research while posing serious environmental concerns. Yet, progress on model efficiency has been impeded by practical challenges in model evaluation and comparison. For example, hardware is challenging to control due to disparate levels of accessibility across different institutions. Moreover, improvements in metrics such as FLOPs often fail to translate to progress in real-world applications. In response, we introduce Pentathlon, a benchmark for holistic and re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.09701","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-07-19T01:05:33Z","cross_cats_sorted":[],"title_canon_sha256":"7ea1774d3c14913dc042acbf171023a4c5c137a2e5e2313dbf568e9113476164","abstract_canon_sha256":"e57313f6aff507c3c3d263f1f70d8caf1487fd3bf49006a90a55b8a81078ee48"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:32:44.734879Z","signature_b64":"8HAqhLNgXnxZVYhdxc+je/qhD1LvTbRN3Ll0p/9QYjYeUE9HSvbJ5kPGQNEEyz960cPi7eYvi4qrBYHPLarkAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c2ed2ea19229336b1f21913e8da7bb50f0887d4fc43e53f1bfbf5a4a591f43a4","last_reissued_at":"2026-07-05T06:32:44.734452Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:32:44.734452Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficiency Pentathlon: A Standardized Arena for Efficiency Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Darrell Plessas, Emma Strubell, Evan Pete Walsh, Hannaneh Hajishirzi, Hao Peng, Iz Beltagy, Jared Fernandez, Jesse Dodge, Kyle Lo, Matthew E. Peters, Noah A. Smith, Qingqing Cao, Sam Skjonsberg, Tom Sherborne","submitted_at":"2023-07-19T01:05:33Z","abstract_excerpt":"Rising computational demands of modern natural language processing (NLP) systems have increased the barrier to entry for cutting-edge research while posing serious environmental concerns. Yet, progress on model efficiency has been impeded by practical challenges in model evaluation and comparison. For example, hardware is challenging to control due to disparate levels of accessibility across different institutions. Moreover, improvements in metrics such as FLOPs often fail to translate to progress in real-world applications. In response, we introduce Pentathlon, a benchmark for holistic and re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.09701","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.09701/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.09701","created_at":"2026-07-05T06:32:44.734515+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.09701v1","created_at":"2026-07-05T06:32:44.734515+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.09701","created_at":"2026-07-05T06:32:44.734515+00:00"},{"alias_kind":"pith_short_12","alias_value":"YLWS5IMSFEZW","created_at":"2026-07-05T06:32:44.734515+00:00"},{"alias_kind":"pith_short_16","alias_value":"YLWS5IMSFEZWWHZB","created_at":"2026-07-05T06:32:44.734515+00:00"},{"alias_kind":"pith_short_8","alias_value":"YLWS5IMS","created_at":"2026-07-05T06:32:44.734515+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YLWS5IMSFEZWWHZBSE7I3J53KD","json":"https://pith.science/pith/YLWS5IMSFEZWWHZBSE7I3J53KD.json","graph_json":"https://pith.science/api/pith-number/YLWS5IMSFEZWWHZBSE7I3J53KD/graph.json","events_json":"https://pith.science/api/pith-number/YLWS5IMSFEZWWHZBSE7I3J53KD/events.json","paper":"https://pith.science/paper/YLWS5IMS"},"agent_actions":{"view_html":"https://pith.science/pith/YLWS5IMSFEZWWHZBSE7I3J53KD","download_json":"https://pith.science/pith/YLWS5IMSFEZWWHZBSE7I3J53KD.json","view_paper":"https://pith.science/paper/YLWS5IMS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.09701&json=true","fetch_graph":"https://pith.science/api/pith-number/YLWS5IMSFEZWWHZBSE7I3J53KD/graph.json","fetch_events":"https://pith.science/api/pith-number/YLWS5IMSFEZWWHZBSE7I3J53KD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YLWS5IMSFEZWWHZBSE7I3J53KD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YLWS5IMSFEZWWHZBSE7I3J53KD/action/storage_attestation","attest_author":"https://pith.science/pith/YLWS5IMSFEZWWHZBSE7I3J53KD/action/author_attestation","sign_citation":"https://pith.science/pith/YLWS5IMSFEZWWHZBSE7I3J53KD/action/citation_signature","submit_replication":"https://pith.science/pith/YLWS5IMSFEZWWHZBSE7I3J53KD/action/replication_record"}},"created_at":"2026-07-05T06:32:44.734515+00:00","updated_at":"2026-07-05T06:32:44.734515+00:00"}