{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:V5Q2CFGC7H6C5OOPZT7AGLLEP3","short_pith_number":"pith:V5Q2CFGC","schema_version":"1.0","canonical_sha256":"af61a114c2f9fc2eb9cfccfe032d647ecbc35ac5e09afccd5989ae2f51b87de1","source":{"kind":"arxiv","id":"2310.19240","version":2},"attestation_state":"computed","paper":{"title":"M4LE: A Multi-Ability Multi-Range Multi-Task Multi-Domain Long-Context Evaluation Benchmark for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kam-Fai Wong, Liangyou Li, Lifeng Shang, Qun Liu, Wai-Chung Kwan, Xingshan Zeng, Yufei Wang, Yusen Sun","submitted_at":"2023-10-30T03:11:30Z","abstract_excerpt":"Managing long sequences has become an important and necessary feature for large language models (LLMs). However, it is still an open question of how to comprehensively and systematically evaluate the long-sequence capability of LLMs. One of the reasons is that conventional and widely-used benchmarks mainly consist of short sequences. In this paper, we propose M4LE, a Multi-ability, Multi-range, Multi-task, Multi-domain benchmark for Long-context Evaluation. M4LE is based on a diverse NLP task pool comprising 36 NLP datasets, 11 task types and 12 domains. To alleviate the scarcity of tasks with"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.19240","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-30T03:11:30Z","cross_cats_sorted":[],"title_canon_sha256":"c2592b6498c12ef32066055e16eb2de4c5a4abdcb421a28899c342bf5212a70f","abstract_canon_sha256":"c26518386218df6d3b69e6951dd41a7656da33458017c98efdf388b9f41718b3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:48:58.556461Z","signature_b64":"zz/tzSgCRSQ3exfgpMgZIxTuG5UkznPQBwzl/psUjEjwyEMhhsh+lmlv1CoJG5KEf2PSt4tOTFMw/kJPIJGlAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af61a114c2f9fc2eb9cfccfe032d647ecbc35ac5e09afccd5989ae2f51b87de1","last_reissued_at":"2026-07-05T08:48:58.556092Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:48:58.556092Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"M4LE: A Multi-Ability Multi-Range Multi-Task Multi-Domain Long-Context Evaluation Benchmark for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kam-Fai Wong, Liangyou Li, Lifeng Shang, Qun Liu, Wai-Chung Kwan, Xingshan Zeng, Yufei Wang, Yusen Sun","submitted_at":"2023-10-30T03:11:30Z","abstract_excerpt":"Managing long sequences has become an important and necessary feature for large language models (LLMs). However, it is still an open question of how to comprehensively and systematically evaluate the long-sequence capability of LLMs. One of the reasons is that conventional and widely-used benchmarks mainly consist of short sequences. In this paper, we propose M4LE, a Multi-ability, Multi-range, Multi-task, Multi-domain benchmark for Long-context Evaluation. M4LE is based on a diverse NLP task pool comprising 36 NLP datasets, 11 task types and 12 domains. To alleviate the scarcity of tasks with"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.19240","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.19240/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.19240","created_at":"2026-07-05T08:48:58.556150+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.19240v2","created_at":"2026-07-05T08:48:58.556150+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.19240","created_at":"2026-07-05T08:48:58.556150+00:00"},{"alias_kind":"pith_short_12","alias_value":"V5Q2CFGC7H6C","created_at":"2026-07-05T08:48:58.556150+00:00"},{"alias_kind":"pith_short_16","alias_value":"V5Q2CFGC7H6C5OOP","created_at":"2026-07-05T08:48:58.556150+00:00"},{"alias_kind":"pith_short_8","alias_value":"V5Q2CFGC","created_at":"2026-07-05T08:48:58.556150+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.19442","citing_title":"A Survey on Large Language Model Acceleration based on KV Cache Management","ref_index":286,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V5Q2CFGC7H6C5OOPZT7AGLLEP3","json":"https://pith.science/pith/V5Q2CFGC7H6C5OOPZT7AGLLEP3.json","graph_json":"https://pith.science/api/pith-number/V5Q2CFGC7H6C5OOPZT7AGLLEP3/graph.json","events_json":"https://pith.science/api/pith-number/V5Q2CFGC7H6C5OOPZT7AGLLEP3/events.json","paper":"https://pith.science/paper/V5Q2CFGC"},"agent_actions":{"view_html":"https://pith.science/pith/V5Q2CFGC7H6C5OOPZT7AGLLEP3","download_json":"https://pith.science/pith/V5Q2CFGC7H6C5OOPZT7AGLLEP3.json","view_paper":"https://pith.science/paper/V5Q2CFGC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.19240&json=true","fetch_graph":"https://pith.science/api/pith-number/V5Q2CFGC7H6C5OOPZT7AGLLEP3/graph.json","fetch_events":"https://pith.science/api/pith-number/V5Q2CFGC7H6C5OOPZT7AGLLEP3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V5Q2CFGC7H6C5OOPZT7AGLLEP3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V5Q2CFGC7H6C5OOPZT7AGLLEP3/action/storage_attestation","attest_author":"https://pith.science/pith/V5Q2CFGC7H6C5OOPZT7AGLLEP3/action/author_attestation","sign_citation":"https://pith.science/pith/V5Q2CFGC7H6C5OOPZT7AGLLEP3/action/citation_signature","submit_replication":"https://pith.science/pith/V5Q2CFGC7H6C5OOPZT7AGLLEP3/action/replication_record"}},"created_at":"2026-07-05T08:48:58.556150+00:00","updated_at":"2026-07-05T08:48:58.556150+00:00"}