{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UWT5ZSE7N4VTO3DH5653VXVM2Y","short_pith_number":"pith:UWT5ZSE7","schema_version":"1.0","canonical_sha256":"a5a7dcc89f6f2b376c67efbbbadeacd639a5eb83b45a70e671fca992c1f597cd","source":{"kind":"arxiv","id":"2502.14317","version":2},"attestation_state":"computed","paper":{"title":"ParallelComp: Parallel Long-Context Compressor for Length Extrapolation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chenyang Zhao, Chiwun Yang, Chuanyang Zheng, Fanghua Ye, Hongxia Yang, Jianghan Shen, Jing Xiong, Lingpeng Kong, Ngai Wong, Zhongwei Wan","submitted_at":"2025-02-20T07:10:43Z","abstract_excerpt":"Extrapolating ultra-long contexts (text length >128K) remains a major challenge for large language models (LLMs), as most training-free extrapolation methods are not only severely limited by memory bottlenecks, but also suffer from the attention sink, which restricts their scalability and effectiveness in practice. In this work, we propose ParallelComp, a parallel long-context compression method that effectively overcomes the memory bottleneck, enabling 8B-parameter LLMs to extrapolate from 8K to 128K tokens on a single A100 80GB GPU in a training-free setting. ParallelComp splits the input in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.14317","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-20T07:10:43Z","cross_cats_sorted":[],"title_canon_sha256":"13cc2825f048477ea07bb554210bbd573a27193d0d56e5feb4a1849e8252abf7","abstract_canon_sha256":"7fe41c94fbdbefd3a17108c5e4bca5df8f7d72613a0c2ba31ecfa5305fa2935b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:18.253615Z","signature_b64":"qx9F2Knn0IhLohwH/Qo24k9ps/vbcAJqQO4CjrsDD2BZsIH/wWNhFuVtHs6XKanKVxecCqOAa3KZsKBnl/y6Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a5a7dcc89f6f2b376c67efbbbadeacd639a5eb83b45a70e671fca992c1f597cd","last_reissued_at":"2026-07-05T11:18:18.253121Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:18.253121Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ParallelComp: Parallel Long-Context Compressor for Length Extrapolation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chenyang Zhao, Chiwun Yang, Chuanyang Zheng, Fanghua Ye, Hongxia Yang, Jianghan Shen, Jing Xiong, Lingpeng Kong, Ngai Wong, Zhongwei Wan","submitted_at":"2025-02-20T07:10:43Z","abstract_excerpt":"Extrapolating ultra-long contexts (text length >128K) remains a major challenge for large language models (LLMs), as most training-free extrapolation methods are not only severely limited by memory bottlenecks, but also suffer from the attention sink, which restricts their scalability and effectiveness in practice. In this work, we propose ParallelComp, a parallel long-context compression method that effectively overcomes the memory bottleneck, enabling 8B-parameter LLMs to extrapolate from 8K to 128K tokens on a single A100 80GB GPU in a training-free setting. ParallelComp splits the input in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.14317","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.14317/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.14317","created_at":"2026-07-05T11:18:18.253179+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.14317v2","created_at":"2026-07-05T11:18:18.253179+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.14317","created_at":"2026-07-05T11:18:18.253179+00:00"},{"alias_kind":"pith_short_12","alias_value":"UWT5ZSE7N4VT","created_at":"2026-07-05T11:18:18.253179+00:00"},{"alias_kind":"pith_short_16","alias_value":"UWT5ZSE7N4VTO3DH","created_at":"2026-07-05T11:18:18.253179+00:00"},{"alias_kind":"pith_short_8","alias_value":"UWT5ZSE7","created_at":"2026-07-05T11:18:18.253179+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.10718","citing_title":"SnapMLA: Efficient Long-Context MLA Decoding via Hardware-Aware FP8 Quantized Pipelining","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10235","citing_title":"CodeComp: Structural KV Cache Compression for Agentic Coding","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UWT5ZSE7N4VTO3DH5653VXVM2Y","json":"https://pith.science/pith/UWT5ZSE7N4VTO3DH5653VXVM2Y.json","graph_json":"https://pith.science/api/pith-number/UWT5ZSE7N4VTO3DH5653VXVM2Y/graph.json","events_json":"https://pith.science/api/pith-number/UWT5ZSE7N4VTO3DH5653VXVM2Y/events.json","paper":"https://pith.science/paper/UWT5ZSE7"},"agent_actions":{"view_html":"https://pith.science/pith/UWT5ZSE7N4VTO3DH5653VXVM2Y","download_json":"https://pith.science/pith/UWT5ZSE7N4VTO3DH5653VXVM2Y.json","view_paper":"https://pith.science/paper/UWT5ZSE7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.14317&json=true","fetch_graph":"https://pith.science/api/pith-number/UWT5ZSE7N4VTO3DH5653VXVM2Y/graph.json","fetch_events":"https://pith.science/api/pith-number/UWT5ZSE7N4VTO3DH5653VXVM2Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UWT5ZSE7N4VTO3DH5653VXVM2Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UWT5ZSE7N4VTO3DH5653VXVM2Y/action/storage_attestation","attest_author":"https://pith.science/pith/UWT5ZSE7N4VTO3DH5653VXVM2Y/action/author_attestation","sign_citation":"https://pith.science/pith/UWT5ZSE7N4VTO3DH5653VXVM2Y/action/citation_signature","submit_replication":"https://pith.science/pith/UWT5ZSE7N4VTO3DH5653VXVM2Y/action/replication_record"}},"created_at":"2026-07-05T11:18:18.253179+00:00","updated_at":"2026-07-05T11:18:18.253179+00:00"}