{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:32RKI6PKEJQLX24AVUNI7FOE5V","short_pith_number":"pith:32RKI6PK","schema_version":"1.0","canonical_sha256":"dea2a479ea2260bbeb80ad1a8f95c4ed40cd7c77c0f428f6e8c519f9c3302d8d","source":{"kind":"arxiv","id":"2506.19651","version":3},"attestation_state":"computed","paper":{"title":"PEVLM: Parallel Encoding for Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.PF"],"primary_cat":"cs.CV","authors_text":"Jin Yang, Letian Kang, Shenxuan Zhou, Shixian Luo, Xiaoyang Yu, Yiqiang Li, Yong Wu, Yuxin Yin","submitted_at":"2025-06-24T14:14:52Z","abstract_excerpt":"Vision-Language Models (VLMs) have demonstrated strong capabilities in multimodal understanding and generation tasks. However, their application to long video understanding remains hindered by the quadratic complexity of standard attention mechanisms. In this work, we introduce \\textbf{PEVLM}, a fine-tuning-free parallel encoding method designed to enhance the prefilling efficiency of VLMs in long video scenarios. PEVLM partitions the input video into context blocks with a shared sink block, while preserving sequential position embeddings to align the attention weight distribution with that of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.19651","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-24T14:14:52Z","cross_cats_sorted":["cs.LG","cs.PF"],"title_canon_sha256":"8688fc5ac4a374b2085ff3517964274796da180699e41094e4bfa7553376de99","abstract_canon_sha256":"0907e9c9604a96fd2c52fd47a60524736152d363dbcf9e39323cb184940fd8ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:44:50.361861Z","signature_b64":"s/9MEpAqTH+j7kCdkdUHKryHI/EnDP0LyJ3Qhv8kKK5ZwF+fig7urMkA0KvCUAfX5c23NKJYZwxsSqtHCExyCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dea2a479ea2260bbeb80ad1a8f95c4ed40cd7c77c0f428f6e8c519f9c3302d8d","last_reissued_at":"2026-07-05T11:44:50.361387Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:44:50.361387Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PEVLM: Parallel Encoding for Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.PF"],"primary_cat":"cs.CV","authors_text":"Jin Yang, Letian Kang, Shenxuan Zhou, Shixian Luo, Xiaoyang Yu, Yiqiang Li, Yong Wu, Yuxin Yin","submitted_at":"2025-06-24T14:14:52Z","abstract_excerpt":"Vision-Language Models (VLMs) have demonstrated strong capabilities in multimodal understanding and generation tasks. However, their application to long video understanding remains hindered by the quadratic complexity of standard attention mechanisms. In this work, we introduce \\textbf{PEVLM}, a fine-tuning-free parallel encoding method designed to enhance the prefilling efficiency of VLMs in long video scenarios. PEVLM partitions the input video into context blocks with a shared sink block, while preserving sequential position embeddings to align the attention weight distribution with that of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.19651","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.19651/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.19651","created_at":"2026-07-05T11:44:50.361444+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.19651v3","created_at":"2026-07-05T11:44:50.361444+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.19651","created_at":"2026-07-05T11:44:50.361444+00:00"},{"alias_kind":"pith_short_12","alias_value":"32RKI6PKEJQL","created_at":"2026-07-05T11:44:50.361444+00:00"},{"alias_kind":"pith_short_16","alias_value":"32RKI6PKEJQLX24A","created_at":"2026-07-05T11:44:50.361444+00:00"},{"alias_kind":"pith_short_8","alias_value":"32RKI6PK","created_at":"2026-07-05T11:44:50.361444+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.10098","citing_title":"Attention Sink in Transformers: A Survey on Utilization, Interpretation, and Mitigation","ref_index":103,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/32RKI6PKEJQLX24AVUNI7FOE5V","json":"https://pith.science/pith/32RKI6PKEJQLX24AVUNI7FOE5V.json","graph_json":"https://pith.science/api/pith-number/32RKI6PKEJQLX24AVUNI7FOE5V/graph.json","events_json":"https://pith.science/api/pith-number/32RKI6PKEJQLX24AVUNI7FOE5V/events.json","paper":"https://pith.science/paper/32RKI6PK"},"agent_actions":{"view_html":"https://pith.science/pith/32RKI6PKEJQLX24AVUNI7FOE5V","download_json":"https://pith.science/pith/32RKI6PKEJQLX24AVUNI7FOE5V.json","view_paper":"https://pith.science/paper/32RKI6PK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.19651&json=true","fetch_graph":"https://pith.science/api/pith-number/32RKI6PKEJQLX24AVUNI7FOE5V/graph.json","fetch_events":"https://pith.science/api/pith-number/32RKI6PKEJQLX24AVUNI7FOE5V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/32RKI6PKEJQLX24AVUNI7FOE5V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/32RKI6PKEJQLX24AVUNI7FOE5V/action/storage_attestation","attest_author":"https://pith.science/pith/32RKI6PKEJQLX24AVUNI7FOE5V/action/author_attestation","sign_citation":"https://pith.science/pith/32RKI6PKEJQLX24AVUNI7FOE5V/action/citation_signature","submit_replication":"https://pith.science/pith/32RKI6PKEJQLX24AVUNI7FOE5V/action/replication_record"}},"created_at":"2026-07-05T11:44:50.361444+00:00","updated_at":"2026-07-05T11:44:50.361444+00:00"}