{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XLVUEGHLB7O6TO6CR4XI4F2IF6","short_pith_number":"pith:XLVUEGHL","schema_version":"1.0","canonical_sha256":"baeb4218eb0fdde9bbc28f2e8e17482f86607a9f9c80957231a86dd780a6317e","source":{"kind":"arxiv","id":"2412.09616","version":2},"attestation_state":"computed","paper":{"title":"V2PE: Improving Multimodal Long-Context Capability of Vision-Language Models with Variable Visual Position Encoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jifeng Dai, Jinguo Zhu, Jintao Lin, Junqi Ge, Xihui Liu, Xizhou Zhu, Ziyi Chen","submitted_at":"2024-12-12T18:59:46Z","abstract_excerpt":"Vision-Language Models (VLMs) have shown promising capabilities in handling various multimodal tasks, yet they struggle in long-context scenarios, particularly in tasks involving videos, high-resolution images, or lengthy image-text documents. In our work, we first conduct an empirical analysis of the long-context capabilities of VLMs using our augmented long-context multimodal datasets. Our findings reveal that directly applying the positional encoding mechanism used for textual tokens to visual tokens is suboptimal, and VLM performance degrades sharply when the position encoding exceeds the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.09616","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-12T18:59:46Z","cross_cats_sorted":[],"title_canon_sha256":"7e32d43344742147c59ca6d1a86670528cd0bf511af34beb84809c2ae10fece1","abstract_canon_sha256":"44927dc7d963a95e49d7dfd655e4264875de4ca4ddd6cb00bce25e6a46693a4f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:48:39.114697Z","signature_b64":"YOValGsOTe+K1dEYhzt4rmMa/uReH0waILZp1S9n9XDStW+FbjpFrcgvSqT5yt1b9mTT99pbNggVp5DWnJVBBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"baeb4218eb0fdde9bbc28f2e8e17482f86607a9f9c80957231a86dd780a6317e","last_reissued_at":"2026-07-05T09:48:39.114215Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:48:39.114215Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"V2PE: Improving Multimodal Long-Context Capability of Vision-Language Models with Variable Visual Position Encoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jifeng Dai, Jinguo Zhu, Jintao Lin, Junqi Ge, Xihui Liu, Xizhou Zhu, Ziyi Chen","submitted_at":"2024-12-12T18:59:46Z","abstract_excerpt":"Vision-Language Models (VLMs) have shown promising capabilities in handling various multimodal tasks, yet they struggle in long-context scenarios, particularly in tasks involving videos, high-resolution images, or lengthy image-text documents. In our work, we first conduct an empirical analysis of the long-context capabilities of VLMs using our augmented long-context multimodal datasets. Our findings reveal that directly applying the positional encoding mechanism used for textual tokens to visual tokens is suboptimal, and VLM performance degrades sharply when the position encoding exceeds the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.09616","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.09616/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.09616","created_at":"2026-07-05T09:48:39.114273+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.09616v2","created_at":"2026-07-05T09:48:39.114273+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.09616","created_at":"2026-07-05T09:48:39.114273+00:00"},{"alias_kind":"pith_short_12","alias_value":"XLVUEGHLB7O6","created_at":"2026-07-05T09:48:39.114273+00:00"},{"alias_kind":"pith_short_16","alias_value":"XLVUEGHLB7O6TO6C","created_at":"2026-07-05T09:48:39.114273+00:00"},{"alias_kind":"pith_short_8","alias_value":"XLVUEGHL","created_at":"2026-07-05T09:48:39.114273+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.14493","citing_title":"LingoLoop Attack: Trapping MLLMs via Linguistic Context and State Entrapment into Endless Loops","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2510.22102","citing_title":"Mitigating Coordinate Prediction Bias from Positional Encoding Failures","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02371","citing_title":"Internalized Reasoning for Long-Context Visual Document Understanding","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10479","citing_title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XLVUEGHLB7O6TO6CR4XI4F2IF6","json":"https://pith.science/pith/XLVUEGHLB7O6TO6CR4XI4F2IF6.json","graph_json":"https://pith.science/api/pith-number/XLVUEGHLB7O6TO6CR4XI4F2IF6/graph.json","events_json":"https://pith.science/api/pith-number/XLVUEGHLB7O6TO6CR4XI4F2IF6/events.json","paper":"https://pith.science/paper/XLVUEGHL"},"agent_actions":{"view_html":"https://pith.science/pith/XLVUEGHLB7O6TO6CR4XI4F2IF6","download_json":"https://pith.science/pith/XLVUEGHLB7O6TO6CR4XI4F2IF6.json","view_paper":"https://pith.science/paper/XLVUEGHL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.09616&json=true","fetch_graph":"https://pith.science/api/pith-number/XLVUEGHLB7O6TO6CR4XI4F2IF6/graph.json","fetch_events":"https://pith.science/api/pith-number/XLVUEGHLB7O6TO6CR4XI4F2IF6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XLVUEGHLB7O6TO6CR4XI4F2IF6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XLVUEGHLB7O6TO6CR4XI4F2IF6/action/storage_attestation","attest_author":"https://pith.science/pith/XLVUEGHLB7O6TO6CR4XI4F2IF6/action/author_attestation","sign_citation":"https://pith.science/pith/XLVUEGHLB7O6TO6CR4XI4F2IF6/action/citation_signature","submit_replication":"https://pith.science/pith/XLVUEGHLB7O6TO6CR4XI4F2IF6/action/replication_record"}},"created_at":"2026-07-05T09:48:39.114273+00:00","updated_at":"2026-07-05T09:48:39.114273+00:00"}