{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:N22FAKQCEWMO7LY3UZFAZ7VY7Q","short_pith_number":"pith:N22FAKQC","canonical_record":{"source":{"id":"2202.03799","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-02-08T11:44:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"740992d08f52820712635a2e8fbc997580580527c93c72f21707a8427461fa90","abstract_canon_sha256":"bfb4d8de44a21d64b9a8a332e4475dd9b7358179e66a57652efe67d7ee8cc3d6"},"schema_version":"1.0"},"canonical_sha256":"6eb4502a022598efaf1ba64a0cfeb8fc2982c0f5dd6c99576f9865949384f355","source":{"kind":"arxiv","id":"2202.03799","version":4},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2202.03799","created_at":"2026-07-05T05:04:12Z"},{"alias_kind":"arxiv_version","alias_value":"2202.03799v4","created_at":"2026-07-05T05:04:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.03799","created_at":"2026-07-05T05:04:12Z"},{"alias_kind":"pith_short_12","alias_value":"N22FAKQCEWMO","created_at":"2026-07-05T05:04:12Z"},{"alias_kind":"pith_short_16","alias_value":"N22FAKQCEWMO7LY3","created_at":"2026-07-05T05:04:12Z"},{"alias_kind":"pith_short_8","alias_value":"N22FAKQC","created_at":"2026-07-05T05:04:12Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:N22FAKQCEWMO7LY3UZFAZ7VY7Q","target":"record","payload":{"canonical_record":{"source":{"id":"2202.03799","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-02-08T11:44:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"740992d08f52820712635a2e8fbc997580580527c93c72f21707a8427461fa90","abstract_canon_sha256":"bfb4d8de44a21d64b9a8a332e4475dd9b7358179e66a57652efe67d7ee8cc3d6"},"schema_version":"1.0"},"canonical_sha256":"6eb4502a022598efaf1ba64a0cfeb8fc2982c0f5dd6c99576f9865949384f355","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:04:12.040022Z","signature_b64":"eZKwVMQV3W6E8H6dfQ3J+eTT/OqBEZAnDWyx9MQv6rOplWvlTz3bSP2kp25tYH+Ou8A+kpN0KqggqKOutdaNCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6eb4502a022598efaf1ba64a0cfeb8fc2982c0f5dd6c99576f9865949384f355","last_reissued_at":"2026-07-05T05:04:12.039570Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:04:12.039570Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2202.03799","source_version":4,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:04:12Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"NrM54EvuxXRyYSsQH0L7uUUr6ETpYf75aF6BqQqtud110rSMmlRm46RvNHOtL/QT4U4ipV55EYWxElVNPqWBBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-22T00:10:43.077182Z"},"content_sha256":"998188654157ff20874d5b4e3017d4fdab9dbcf9ae74664ce20c5a6d0145f308","schema_version":"1.0","event_id":"sha256:998188654157ff20874d5b4e3017d4fdab9dbcf9ae74664ce20c5a6d0145f308"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:N22FAKQCEWMO7LY3UZFAZ7VY7Q","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"What are the best systems? New perspectives on NLP Benchmarking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ekhine Irurozki, Nathan Noiry, Pierre Colombo, Stephan Clemencon","submitted_at":"2022-02-08T11:44:20Z","abstract_excerpt":"In Machine Learning, a benchmark refers to an ensemble of datasets associated with one or multiple metrics together with a way to aggregate different systems performances. They are instrumental in (i) assessing the progress of new methods along different axes and (ii) selecting the best systems for practical use. This is particularly the case for NLP with the development of large pre-trained models (e.g. GPT, BERT) that are expected to generalize well on a variety of tasks. While the community mainly focused on developing new datasets and metrics, there has been little interest in the aggregat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.03799","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.03799/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:04:12Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"lKs/HX8a4NWJDYHo0XWp0eLAVNfdDXY8fde+rQPep+Nt5Juv8TlXcCxaT6wUQoGKDqc+LIVE2gganuWy6s00CQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-22T00:10:43.077872Z"},"content_sha256":"4fbb2649c7c10356c647a6a8c166a63fc6a080f1790fda6eb7dbd490553e7890","schema_version":"1.0","event_id":"sha256:4fbb2649c7c10356c647a6a8c166a63fc6a080f1790fda6eb7dbd490553e7890"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/N22FAKQCEWMO7LY3UZFAZ7VY7Q/bundle.json","state_url":"https://pith.science/pith/N22FAKQCEWMO7LY3UZFAZ7VY7Q/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/N22FAKQCEWMO7LY3UZFAZ7VY7Q/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-22T00:10:43Z","links":{"resolver":"https://pith.science/pith/N22FAKQCEWMO7LY3UZFAZ7VY7Q","bundle":"https://pith.science/pith/N22FAKQCEWMO7LY3UZFAZ7VY7Q/bundle.json","state":"https://pith.science/pith/N22FAKQCEWMO7LY3UZFAZ7VY7Q/state.json","well_known_bundle":"https://pith.science/.well-known/pith/N22FAKQCEWMO7LY3UZFAZ7VY7Q/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:N22FAKQCEWMO7LY3UZFAZ7VY7Q","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"bfb4d8de44a21d64b9a8a332e4475dd9b7358179e66a57652efe67d7ee8cc3d6","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-02-08T11:44:20Z","title_canon_sha256":"740992d08f52820712635a2e8fbc997580580527c93c72f21707a8427461fa90"},"schema_version":"1.0","source":{"id":"2202.03799","kind":"arxiv","version":4}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2202.03799","created_at":"2026-07-05T05:04:12Z"},{"alias_kind":"arxiv_version","alias_value":"2202.03799v4","created_at":"2026-07-05T05:04:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.03799","created_at":"2026-07-05T05:04:12Z"},{"alias_kind":"pith_short_12","alias_value":"N22FAKQCEWMO","created_at":"2026-07-05T05:04:12Z"},{"alias_kind":"pith_short_16","alias_value":"N22FAKQCEWMO7LY3","created_at":"2026-07-05T05:04:12Z"},{"alias_kind":"pith_short_8","alias_value":"N22FAKQC","created_at":"2026-07-05T05:04:12Z"}],"graph_snapshots":[{"event_id":"sha256:4fbb2649c7c10356c647a6a8c166a63fc6a080f1790fda6eb7dbd490553e7890","target":"graph","created_at":"2026-07-05T05:04:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2202.03799/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In Machine Learning, a benchmark refers to an ensemble of datasets associated with one or multiple metrics together with a way to aggregate different systems performances. They are instrumental in (i) assessing the progress of new methods along different axes and (ii) selecting the best systems for practical use. This is particularly the case for NLP with the development of large pre-trained models (e.g. GPT, BERT) that are expected to generalize well on a variety of tasks. While the community mainly focused on developing new datasets and metrics, there has been little interest in the aggregat","authors_text":"Ekhine Irurozki, Nathan Noiry, Pierre Colombo, Stephan Clemencon","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-02-08T11:44:20Z","title":"What are the best systems? New perspectives on NLP Benchmarking"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.03799","kind":"arxiv","version":4},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:998188654157ff20874d5b4e3017d4fdab9dbcf9ae74664ce20c5a6d0145f308","target":"record","created_at":"2026-07-05T05:04:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"bfb4d8de44a21d64b9a8a332e4475dd9b7358179e66a57652efe67d7ee8cc3d6","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-02-08T11:44:20Z","title_canon_sha256":"740992d08f52820712635a2e8fbc997580580527c93c72f21707a8427461fa90"},"schema_version":"1.0","source":{"id":"2202.03799","kind":"arxiv","version":4}},"canonical_sha256":"6eb4502a022598efaf1ba64a0cfeb8fc2982c0f5dd6c99576f9865949384f355","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6eb4502a022598efaf1ba64a0cfeb8fc2982c0f5dd6c99576f9865949384f355","first_computed_at":"2026-07-05T05:04:12.039570Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:04:12.039570Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"eZKwVMQV3W6E8H6dfQ3J+eTT/OqBEZAnDWyx9MQv6rOplWvlTz3bSP2kp25tYH+Ou8A+kpN0KqggqKOutdaNCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T05:04:12.040022Z","signed_message":"canonical_sha256_bytes"},"source_id":"2202.03799","source_kind":"arxiv","source_version":4}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:998188654157ff20874d5b4e3017d4fdab9dbcf9ae74664ce20c5a6d0145f308","sha256:4fbb2649c7c10356c647a6a8c166a63fc6a080f1790fda6eb7dbd490553e7890"],"state_sha256":"5c81f6bf8493af951e66ea940b8a8dbb92a4f944e960e4e967e8931c59d7bc3a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"mLpd3WnD49UmWFoxY3UnQ6Z9ZHEoD+6zzd2DbVA+4F2AqyZVO0G7IjAzxd3tYrCoOvoo/drLos4dpPuNUFd4CQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-22T00:10:43.082376Z","bundle_sha256":"634fcedd4eee2b3fcb3a094da5d72f516f21d2b8572ed7c43ca28770b899c266"}}