{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:LG2UJS67U7FWR6PFGLSES64YJJ","short_pith_number":"pith:LG2UJS67","canonical_record":{"source":{"id":"2406.16852","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-24T17:58:06Z","cross_cats_sorted":[],"title_canon_sha256":"7f980a30439ebd4cf7380e71e67a01edbee0f0973af5c5e550e75bf9799b754d","abstract_canon_sha256":"22fec4458a2cce994c21ab7fbdc6a281edf3b830207eb5c5a5c2d79222e2d509"},"schema_version":"1.0"},"canonical_sha256":"59b544cbdfa7cb68f9e532e4497b984a5cf650cfe9a2bf69e419817d74723cc2","source":{"kind":"arxiv","id":"2406.16852","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.16852","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"arxiv_version","alias_value":"2406.16852v2","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16852","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"pith_short_12","alias_value":"LG2UJS67U7FW","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"pith_short_16","alias_value":"LG2UJS67U7FWR6PF","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"pith_short_8","alias_value":"LG2UJS67","created_at":"2026-07-05T08:38:11Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:LG2UJS67U7FWR6PFGLSES64YJJ","target":"record","payload":{"canonical_record":{"source":{"id":"2406.16852","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-24T17:58:06Z","cross_cats_sorted":[],"title_canon_sha256":"7f980a30439ebd4cf7380e71e67a01edbee0f0973af5c5e550e75bf9799b754d","abstract_canon_sha256":"22fec4458a2cce994c21ab7fbdc6a281edf3b830207eb5c5a5c2d79222e2d509"},"schema_version":"1.0"},"canonical_sha256":"59b544cbdfa7cb68f9e532e4497b984a5cf650cfe9a2bf69e419817d74723cc2","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:11.816820Z","signature_b64":"7nItncez0zIKV+K8Kl6UoRjJWftlbybppkJ92ZYnv0Wfv+PkkUHiMiD4d44zCB9+a2Nsa/yLWD18h953GDxDAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"59b544cbdfa7cb68f9e532e4497b984a5cf650cfe9a2bf69e419817d74723cc2","last_reissued_at":"2026-07-05T08:38:11.816288Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:11.816288Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2406.16852","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:38:11Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"cDrq+4ST16/N5vKvloJHclIaLXzDgfO4x6yANFO3Jmq+ok5wLSBJPD5ssS2L40pPL8Sx6KU0VCSqCxu8J/4VCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T02:06:13.264134Z"},"content_sha256":"4a8b2545a422eceed5ec78c6a56a20e6b0be51b4cd9e0054d8d4f2192af6192c","schema_version":"1.0","event_id":"sha256:4a8b2545a422eceed5ec78c6a56a20e6b0be51b4cd9e0054d8d4f2192af6192c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:LG2UJS67U7FWR6PFGLSES64YJJ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Long Context Transfer from Language to Vision","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Extending language model context length transfers directly to vision, letting multimodal models handle orders of magnitude more visual tokens without video training.","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Li, Chunyuan Li, Guangtao Zeng, Haoran Tan, Jingkang Yang, Kaichen Zhang, Peiyuan Zhang, Yuanhan Zhang, Ziwei Liu, Ziyue Wang","submitted_at":"2024-06-24T17:58:06Z","abstract_excerpt":"Video sequences offer valuable temporal information, but existing large multimodal models (LMMs) fall short in understanding extremely long videos. Many works address this by reducing the number of visual tokens using visual resamplers. Alternatively, in this paper, we approach this problem from the perspective of the language model. By simply extrapolating the context length of the language backbone, we enable LMMs to comprehend orders of magnitude more visual tokens without any video training. We call this phenomenon long context transfer and carefully ablate its properties. To effectively m"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"By simply extrapolating the context length of the language backbone, we enable LMMs to comprehend orders of magnitude more visual tokens without any video training.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That long-context capabilities learned purely from text sequences transfer effectively and without significant degradation to sequences of visual tokens in the same transformer architecture.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Extending language model context length enables LMMs to process over 200K visual tokens from long videos without video training, achieving SOTA on Video-MME via dense frame sampling.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Extending language model context length transfers directly to vision, letting multimodal models handle orders of magnitude more visual tokens without video training.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"16f653f94cfb57aabf13e100acd4221ce755a32286d648a0f2bad29ce9a7fb9a"},"source":{"id":"2406.16852","kind":"arxiv","version":2},"verdict":{"id":"05f39201-d8ee-415d-bb31-f120aad7f005","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-12T07:00:08.278129Z","strongest_claim":"By simply extrapolating the context length of the language backbone, we enable LMMs to comprehend orders of magnitude more visual tokens without any video training.","one_line_summary":"Extending language model context length enables LMMs to process over 200K visual tokens from long videos without video training, achieving SOTA on Video-MME via dense frame sampling.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That long-context capabilities learned purely from text sequences transfer effectively and without significant degradation to sequences of visual tokens in the same transformer architecture.","pith_extraction_headline":"Extending language model context length transfers directly to vision, letting multimodal models handle orders of magnitude more visual tokens without video training."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.16852/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":94,"sample":[{"doi":"","year":2023,"title":"Llm testneedleinahaystack","work_id":"dcd6606b-a611-488d-9f88-ac500351c4ce","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2022,"title":"Flamingo: a visual language model for few-shot learning","work_id":"d0eddd17-3111-48c2-b719-dfad065feac7","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2015,"title":"Vqa: Visual question answering","work_id":"e8f56797-d0b6-4240-9cef-96598039af3b","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2023,"title":"OpenFlamingo: An Open-Source Framework for Training Large Autoregressive Vision-Language Models","work_id":"87bfa84a-e663-4165-806f-93ef439d88d0","ref_index":4,"cited_arxiv_id":"2308.01390","is_internal_anchor":true},{"doi":"","year":2024,"title":"Longalign: A recipe for long context alignment of large language models","work_id":"31413e98-8678-4d3a-87d3-12a7dd4c906c","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":94,"snapshot_sha256":"e50ffd6783b419595ad39e89ba8e8458ba2ed3b0fc16d8325abb91f39ee0d288","internal_anchors":5},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"05f39201-d8ee-415d-bb31-f120aad7f005"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:38:11Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"pWGh0A/X7QazIfEzd9387PkvwPc3+2l/MK18HsaWlusRyXxDNerVGaeJtF7N/wAmeo3/WghkT2jirj0mYrqwDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T02:06:13.265181Z"},"content_sha256":"d1dad2e930f6da22517a5a8feafca034b4d7526666da897175eabffb3008b674","schema_version":"1.0","event_id":"sha256:d1dad2e930f6da22517a5a8feafca034b4d7526666da897175eabffb3008b674"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/LG2UJS67U7FWR6PFGLSES64YJJ/bundle.json","state_url":"https://pith.science/pith/LG2UJS67U7FWR6PFGLSES64YJJ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/LG2UJS67U7FWR6PFGLSES64YJJ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-11T02:06:13Z","links":{"resolver":"https://pith.science/pith/LG2UJS67U7FWR6PFGLSES64YJJ","bundle":"https://pith.science/pith/LG2UJS67U7FWR6PFGLSES64YJJ/bundle.json","state":"https://pith.science/pith/LG2UJS67U7FWR6PFGLSES64YJJ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/LG2UJS67U7FWR6PFGLSES64YJJ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:LG2UJS67U7FWR6PFGLSES64YJJ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"22fec4458a2cce994c21ab7fbdc6a281edf3b830207eb5c5a5c2d79222e2d509","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-24T17:58:06Z","title_canon_sha256":"7f980a30439ebd4cf7380e71e67a01edbee0f0973af5c5e550e75bf9799b754d"},"schema_version":"1.0","source":{"id":"2406.16852","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.16852","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"arxiv_version","alias_value":"2406.16852v2","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16852","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"pith_short_12","alias_value":"LG2UJS67U7FW","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"pith_short_16","alias_value":"LG2UJS67U7FWR6PF","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"pith_short_8","alias_value":"LG2UJS67","created_at":"2026-07-05T08:38:11Z"}],"graph_snapshots":[{"event_id":"sha256:d1dad2e930f6da22517a5a8feafca034b4d7526666da897175eabffb3008b674","target":"graph","created_at":"2026-07-05T08:38:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"By simply extrapolating the context length of the language backbone, we enable LMMs to comprehend orders of magnitude more visual tokens without any video training."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That long-context capabilities learned purely from text sequences transfer effectively and without significant degradation to sequences of visual tokens in the same transformer architecture."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Extending language model context length enables LMMs to process over 200K visual tokens from long videos without video training, achieving SOTA on Video-MME via dense frame sampling."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Extending language model context length transfers directly to vision, letting multimodal models handle orders of magnitude more visual tokens without video training."}],"snapshot_sha256":"16f653f94cfb57aabf13e100acd4221ce755a32286d648a0f2bad29ce9a7fb9a"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2406.16852/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Video sequences offer valuable temporal information, but existing large multimodal models (LMMs) fall short in understanding extremely long videos. Many works address this by reducing the number of visual tokens using visual resamplers. Alternatively, in this paper, we approach this problem from the perspective of the language model. By simply extrapolating the context length of the language backbone, we enable LMMs to comprehend orders of magnitude more visual tokens without any video training. We call this phenomenon long context transfer and carefully ablate its properties. To effectively m","authors_text":"Bo Li, Chunyuan Li, Guangtao Zeng, Haoran Tan, Jingkang Yang, Kaichen Zhang, Peiyuan Zhang, Yuanhan Zhang, Ziwei Liu, Ziyue Wang","cross_cats":[],"headline":"Extending language model context length transfers directly to vision, letting multimodal models handle orders of magnitude more visual tokens without video training.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision"},"references":{"count":94,"internal_anchors":5,"resolved_work":94,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":1,"title":"Llm testneedleinahaystack","work_id":"dcd6606b-a611-488d-9f88-ac500351c4ce","year":2023},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"Flamingo: a visual language model for few-shot learning","work_id":"d0eddd17-3111-48c2-b719-dfad065feac7","year":2022},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"Vqa: Visual question answering","work_id":"e8f56797-d0b6-4240-9cef-96598039af3b","year":2015},{"cited_arxiv_id":"2308.01390","doi":"","is_internal_anchor":true,"ref_index":4,"title":"OpenFlamingo: An Open-Source Framework for Training Large Autoregressive Vision-Language Models","work_id":"87bfa84a-e663-4165-806f-93ef439d88d0","year":2023},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"Longalign: A recipe for long context alignment of large language models","work_id":"31413e98-8678-4d3a-87d3-12a7dd4c906c","year":2024}],"snapshot_sha256":"e50ffd6783b419595ad39e89ba8e8458ba2ed3b0fc16d8325abb91f39ee0d288"},"source":{"id":"2406.16852","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-12T07:00:08.278129Z","id":"05f39201-d8ee-415d-bb31-f120aad7f005","model_set":{"reader":"grok-4.3"},"one_line_summary":"Extending language model context length enables LMMs to process over 200K visual tokens from long videos without video training, achieving SOTA on Video-MME via dense frame sampling.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Extending language model context length transfers directly to vision, letting multimodal models handle orders of magnitude more visual tokens without video training.","strongest_claim":"By simply extrapolating the context length of the language backbone, we enable LMMs to comprehend orders of magnitude more visual tokens without any video training.","weakest_assumption":"That long-context capabilities learned purely from text sequences transfer effectively and without significant degradation to sequences of visual tokens in the same transformer architecture."}},"verdict_id":"05f39201-d8ee-415d-bb31-f120aad7f005"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4a8b2545a422eceed5ec78c6a56a20e6b0be51b4cd9e0054d8d4f2192af6192c","target":"record","created_at":"2026-07-05T08:38:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"22fec4458a2cce994c21ab7fbdc6a281edf3b830207eb5c5a5c2d79222e2d509","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-24T17:58:06Z","title_canon_sha256":"7f980a30439ebd4cf7380e71e67a01edbee0f0973af5c5e550e75bf9799b754d"},"schema_version":"1.0","source":{"id":"2406.16852","kind":"arxiv","version":2}},"canonical_sha256":"59b544cbdfa7cb68f9e532e4497b984a5cf650cfe9a2bf69e419817d74723cc2","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"59b544cbdfa7cb68f9e532e4497b984a5cf650cfe9a2bf69e419817d74723cc2","first_computed_at":"2026-07-05T08:38:11.816288Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:38:11.816288Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"7nItncez0zIKV+K8Kl6UoRjJWftlbybppkJ92ZYnv0Wfv+PkkUHiMiD4d44zCB9+a2Nsa/yLWD18h953GDxDAg==","signature_status":"signed_v1","signed_at":"2026-07-05T08:38:11.816820Z","signed_message":"canonical_sha256_bytes"},"source_id":"2406.16852","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4a8b2545a422eceed5ec78c6a56a20e6b0be51b4cd9e0054d8d4f2192af6192c","sha256:d1dad2e930f6da22517a5a8feafca034b4d7526666da897175eabffb3008b674"],"state_sha256":"2f3ae516222d42bb0ae1ae64748d7f59630cf1422b29b34915dfc20dc1936a69"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xtqBVdm9Remi6fSgb1dKuhHt0RdOQEBFa/gfrUGibqWDNTLUjQvpfdXSnjuaQuyHs5owUzR7Q4IwEBpd/SMmDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-11T02:06:13.272451Z","bundle_sha256":"d0371e3cfdc72d185732756dc08826fd1c9cc3a987b9336db53327c28c0064ea"}}