{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Q2SRUNXUCWDWG54TTVYQGJW6U4","short_pith_number":"pith:Q2SRUNXU","schema_version":"1.0","canonical_sha256":"86a51a36f415876377939d710326dea7165252a50718460617276adbbfd1fd42","source":{"kind":"arxiv","id":"2404.05446","version":1},"attestation_state":"computed","paper":{"title":"XL$^2$Bench: A Benchmark for Extremely Long Context Understanding with Long-range Dependencies","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dawei Yin, Hengyi Cai, Piji Li, Shuaiqiang Wang, Xiaochi Wei, Xuanfan Ni","submitted_at":"2024-04-08T12:29:07Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable performance across diverse tasks but are constrained by their small context window sizes. Various efforts have been proposed to expand the context window to accommodate even up to 200K input tokens. Meanwhile, building high-quality benchmarks with much longer text lengths and more demanding tasks to provide comprehensive evaluations is of immense practical interest to facilitate long context understanding research of LLMs. However, prior benchmarks create datasets that ostensibly cater to long-text comprehension by expanding the input o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.05446","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-08T12:29:07Z","cross_cats_sorted":[],"title_canon_sha256":"5f9aa1dfa52aab056030481f4718863c28e695dc8d602d16301b331d01eaf375","abstract_canon_sha256":"434fbb4cfeee620da6dab6ba039181002cedf77085b74e567d31e73fcf7a90d4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:05:31.118690Z","signature_b64":"2muaTgvQpFPakIBBxq+r8abNVmEpauRh+hBFGtYKsbgO23K2C/FcQ/mb6UPnM+f5BafVokiqbS21r117xmebBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86a51a36f415876377939d710326dea7165252a50718460617276adbbfd1fd42","last_reissued_at":"2026-07-05T08:05:31.118347Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:05:31.118347Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"XL$^2$Bench: A Benchmark for Extremely Long Context Understanding with Long-range Dependencies","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dawei Yin, Hengyi Cai, Piji Li, Shuaiqiang Wang, Xiaochi Wei, Xuanfan Ni","submitted_at":"2024-04-08T12:29:07Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable performance across diverse tasks but are constrained by their small context window sizes. Various efforts have been proposed to expand the context window to accommodate even up to 200K input tokens. Meanwhile, building high-quality benchmarks with much longer text lengths and more demanding tasks to provide comprehensive evaluations is of immense practical interest to facilitate long context understanding research of LLMs. However, prior benchmarks create datasets that ostensibly cater to long-text comprehension by expanding the input o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.05446","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.05446/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.05446","created_at":"2026-07-05T08:05:31.118404+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.05446v1","created_at":"2026-07-05T08:05:31.118404+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.05446","created_at":"2026-07-05T08:05:31.118404+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q2SRUNXUCWDW","created_at":"2026-07-05T08:05:31.118404+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q2SRUNXUCWDWG54T","created_at":"2026-07-05T08:05:31.118404+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q2SRUNXU","created_at":"2026-07-05T08:05:31.118404+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.08235","citing_title":"Can AI Validate Science? Benchmarking LLMs for Accurate Scientific Claim $\\rightarrow$ Evidence Reasoning","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q2SRUNXUCWDWG54TTVYQGJW6U4","json":"https://pith.science/pith/Q2SRUNXUCWDWG54TTVYQGJW6U4.json","graph_json":"https://pith.science/api/pith-number/Q2SRUNXUCWDWG54TTVYQGJW6U4/graph.json","events_json":"https://pith.science/api/pith-number/Q2SRUNXUCWDWG54TTVYQGJW6U4/events.json","paper":"https://pith.science/paper/Q2SRUNXU"},"agent_actions":{"view_html":"https://pith.science/pith/Q2SRUNXUCWDWG54TTVYQGJW6U4","download_json":"https://pith.science/pith/Q2SRUNXUCWDWG54TTVYQGJW6U4.json","view_paper":"https://pith.science/paper/Q2SRUNXU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.05446&json=true","fetch_graph":"https://pith.science/api/pith-number/Q2SRUNXUCWDWG54TTVYQGJW6U4/graph.json","fetch_events":"https://pith.science/api/pith-number/Q2SRUNXUCWDWG54TTVYQGJW6U4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q2SRUNXUCWDWG54TTVYQGJW6U4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q2SRUNXUCWDWG54TTVYQGJW6U4/action/storage_attestation","attest_author":"https://pith.science/pith/Q2SRUNXUCWDWG54TTVYQGJW6U4/action/author_attestation","sign_citation":"https://pith.science/pith/Q2SRUNXUCWDWG54TTVYQGJW6U4/action/citation_signature","submit_replication":"https://pith.science/pith/Q2SRUNXUCWDWG54TTVYQGJW6U4/action/replication_record"}},"created_at":"2026-07-05T08:05:31.118404+00:00","updated_at":"2026-07-05T08:05:31.118404+00:00"}