{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VRPZ57BTISEBDMSVV7OQ35KR7Y","short_pith_number":"pith:VRPZ57BT","schema_version":"1.0","canonical_sha256":"ac5f9efc33448811b255afdd0df551fe2db757f9133bcdf3d2102cd72ec65c61","source":{"kind":"arxiv","id":"2402.10685","version":2},"attestation_state":"computed","paper":{"title":"LongHeads: Multi-Head Attention is Secretly a Long Context Processor","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jun Zhao, Qi Zhang, Tao Gui, Tao Ji, Wei He, Xin Zhou, Xuanjing Huang, Yi Lu","submitted_at":"2024-02-16T13:39:34Z","abstract_excerpt":"Large language models (LLMs) have achieved impressive performance in numerous domains but often struggle to process lengthy inputs effectively and efficiently due to limited length generalization and attention's quadratic computational demands. Many sought to mitigate this by restricting the attention window within the pre-trained length. However, these methods introduce new issues such as ignoring the middle context and requiring additional training. To address these problems, we propose LongHeads, a training-free framework that enhances LLM's long context ability by unlocking multi-head atte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10685","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-16T13:39:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b67da08b68169920dca38ed4e570000fdac6d95f47a02f53338d9e600d69b924","abstract_canon_sha256":"4cbd44898df4a93b0cbc2187d825be20ad226a4f5a8651ac6a02173111c2fbe4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:00:01.478944Z","signature_b64":"Y18Q4B0i8FSc7nyimjFL0UEjmamUahg1G5Lhpnne9J6TsUazsohvfIRdoC8fGbaNEd07slit/ZY8STuD+qoWBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac5f9efc33448811b255afdd0df551fe2db757f9133bcdf3d2102cd72ec65c61","last_reissued_at":"2026-07-05T08:00:01.478421Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:00:01.478421Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LongHeads: Multi-Head Attention is Secretly a Long Context Processor","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jun Zhao, Qi Zhang, Tao Gui, Tao Ji, Wei He, Xin Zhou, Xuanjing Huang, Yi Lu","submitted_at":"2024-02-16T13:39:34Z","abstract_excerpt":"Large language models (LLMs) have achieved impressive performance in numerous domains but often struggle to process lengthy inputs effectively and efficiently due to limited length generalization and attention's quadratic computational demands. Many sought to mitigate this by restricting the attention window within the pre-trained length. However, these methods introduce new issues such as ignoring the middle context and requiring additional training. To address these problems, we propose LongHeads, a training-free framework that enhances LLM's long context ability by unlocking multi-head atte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10685","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10685/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10685","created_at":"2026-07-05T08:00:01.478482+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10685v2","created_at":"2026-07-05T08:00:01.478482+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10685","created_at":"2026-07-05T08:00:01.478482+00:00"},{"alias_kind":"pith_short_12","alias_value":"VRPZ57BTISEB","created_at":"2026-07-05T08:00:01.478482+00:00"},{"alias_kind":"pith_short_16","alias_value":"VRPZ57BTISEBDMSV","created_at":"2026-07-05T08:00:01.478482+00:00"},{"alias_kind":"pith_short_8","alias_value":"VRPZ57BT","created_at":"2026-07-05T08:00:01.478482+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.13189","citing_title":"MoBA: Mixture of Block Attention for Long-Context LLMs","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VRPZ57BTISEBDMSVV7OQ35KR7Y","json":"https://pith.science/pith/VRPZ57BTISEBDMSVV7OQ35KR7Y.json","graph_json":"https://pith.science/api/pith-number/VRPZ57BTISEBDMSVV7OQ35KR7Y/graph.json","events_json":"https://pith.science/api/pith-number/VRPZ57BTISEBDMSVV7OQ35KR7Y/events.json","paper":"https://pith.science/paper/VRPZ57BT"},"agent_actions":{"view_html":"https://pith.science/pith/VRPZ57BTISEBDMSVV7OQ35KR7Y","download_json":"https://pith.science/pith/VRPZ57BTISEBDMSVV7OQ35KR7Y.json","view_paper":"https://pith.science/paper/VRPZ57BT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10685&json=true","fetch_graph":"https://pith.science/api/pith-number/VRPZ57BTISEBDMSVV7OQ35KR7Y/graph.json","fetch_events":"https://pith.science/api/pith-number/VRPZ57BTISEBDMSVV7OQ35KR7Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VRPZ57BTISEBDMSVV7OQ35KR7Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VRPZ57BTISEBDMSVV7OQ35KR7Y/action/storage_attestation","attest_author":"https://pith.science/pith/VRPZ57BTISEBDMSVV7OQ35KR7Y/action/author_attestation","sign_citation":"https://pith.science/pith/VRPZ57BTISEBDMSVV7OQ35KR7Y/action/citation_signature","submit_replication":"https://pith.science/pith/VRPZ57BTISEBDMSVV7OQ35KR7Y/action/replication_record"}},"created_at":"2026-07-05T08:00:01.478482+00:00","updated_at":"2026-07-05T08:00:01.478482+00:00"}