{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UP3LW4KDF5ES57AX62VLY2U3BL","short_pith_number":"pith:UP3LW4KD","schema_version":"1.0","canonical_sha256":"a3f6bb71432f492efc17f6aabc6a9b0af45c8a486c2c8709851e116a15560553","source":{"kind":"arxiv","id":"2311.17043","version":1},"attestation_state":"computed","paper":{"title":"LLaMA-VID: An Image is Worth 2 Tokens in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chengyao Wang, Jiaya Jia, Yanwei Li","submitted_at":"2023-11-28T18:53:43Z","abstract_excerpt":"In this work, we present a novel method to tackle the token generation challenge in Vision Language Models (VLMs) for video and image understanding, called LLaMA-VID. Current VLMs, while proficient in tasks like image captioning and visual question answering, face computational burdens when processing long videos due to the excessive visual tokens. LLaMA-VID addresses this issue by representing each frame with two distinct tokens, namely context token and content token. The context token encodes the overall image context based on user input, whereas the content token encapsulates visual cues i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.17043","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-11-28T18:53:43Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"bc3cf1b493ca012651a188f784fe21839ef2f19978274c2d1d47297ea2198394","abstract_canon_sha256":"a1c51d62780ff91da2bb06084ab7dc69d51c9cf61db8823c3524d1dab119ac6a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:17:49.985597Z","signature_b64":"Iac6M3VDHsawonnG1pXejrSvERowDitC7y/PKqwdtAa5az0BDV3h5RNsdRClXI1ARDvRHz3AaHBZU3U/mmOkCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a3f6bb71432f492efc17f6aabc6a9b0af45c8a486c2c8709851e116a15560553","last_reissued_at":"2026-07-05T07:17:49.985004Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:17:49.985004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLaMA-VID: An Image is Worth 2 Tokens in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chengyao Wang, Jiaya Jia, Yanwei Li","submitted_at":"2023-11-28T18:53:43Z","abstract_excerpt":"In this work, we present a novel method to tackle the token generation challenge in Vision Language Models (VLMs) for video and image understanding, called LLaMA-VID. Current VLMs, while proficient in tasks like image captioning and visual question answering, face computational burdens when processing long videos due to the excessive visual tokens. LLaMA-VID addresses this issue by representing each frame with two distinct tokens, namely context token and content token. The context token encodes the overall image context based on user input, whereas the content token encapsulates visual cues i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.17043","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.17043/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.17043","created_at":"2026-07-05T07:17:49.985066+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.17043v1","created_at":"2026-07-05T07:17:49.985066+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.17043","created_at":"2026-07-05T07:17:49.985066+00:00"},{"alias_kind":"pith_short_12","alias_value":"UP3LW4KDF5ES","created_at":"2026-07-05T07:17:49.985066+00:00"},{"alias_kind":"pith_short_16","alias_value":"UP3LW4KDF5ES57AX","created_at":"2026-07-05T07:17:49.985066+00:00"},{"alias_kind":"pith_short_8","alias_value":"UP3LW4KD","created_at":"2026-07-05T07:17:49.985066+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16740","citing_title":"TRACE: Evidence Grounding-Guided Multi-Video Event Understanding and Claim Generation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2407.08101","citing_title":"What to Say and When to Say it: Live Fitness Coaching as a Testbed for Situated Interaction","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19950","citing_title":"AffectVerse: Emotional World Models for Multimodal Affective Computing","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16740","citing_title":"TRACE: Evidence Grounding-Guided Multi-Video Event Understanding and Claim Generation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2406.08035","citing_title":"LVBench: An Extreme Long Video Understanding Benchmark","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21420","citing_title":"ReGATE: Learning Faster and Better with Fewer Tokens in MLLMs","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2402.15852","citing_title":"NaVid: Video-based VLM Plans the Next Step for Vision-and-Language Navigation","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2407.03320","citing_title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2403.18814","citing_title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2408.10188","citing_title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2403.00476","citing_title":"TempCompass: Do Video LLMs Really Understand Videos?","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2412.06224","citing_title":"Uni-NaVid: A Video-based Vision-Language-Action Model for Unifying Embodied Navigation Tasks","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16994","citing_title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27259","citing_title":"Seeing the Scene Matters: Revealing Forgetting in Video Understanding Models with a Scene-Aware Long-Video Benchmark","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2406.04264","citing_title":"MLVU: Benchmarking Multi-task Long Video Understanding","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2407.07895","citing_title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2410.02713","citing_title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","ref_index":193,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UP3LW4KDF5ES57AX62VLY2U3BL","json":"https://pith.science/pith/UP3LW4KDF5ES57AX62VLY2U3BL.json","graph_json":"https://pith.science/api/pith-number/UP3LW4KDF5ES57AX62VLY2U3BL/graph.json","events_json":"https://pith.science/api/pith-number/UP3LW4KDF5ES57AX62VLY2U3BL/events.json","paper":"https://pith.science/paper/UP3LW4KD"},"agent_actions":{"view_html":"https://pith.science/pith/UP3LW4KDF5ES57AX62VLY2U3BL","download_json":"https://pith.science/pith/UP3LW4KDF5ES57AX62VLY2U3BL.json","view_paper":"https://pith.science/paper/UP3LW4KD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.17043&json=true","fetch_graph":"https://pith.science/api/pith-number/UP3LW4KDF5ES57AX62VLY2U3BL/graph.json","fetch_events":"https://pith.science/api/pith-number/UP3LW4KDF5ES57AX62VLY2U3BL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UP3LW4KDF5ES57AX62VLY2U3BL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UP3LW4KDF5ES57AX62VLY2U3BL/action/storage_attestation","attest_author":"https://pith.science/pith/UP3LW4KDF5ES57AX62VLY2U3BL/action/author_attestation","sign_citation":"https://pith.science/pith/UP3LW4KDF5ES57AX62VLY2U3BL/action/citation_signature","submit_replication":"https://pith.science/pith/UP3LW4KDF5ES57AX62VLY2U3BL/action/replication_record"}},"created_at":"2026-07-05T07:17:49.985066+00:00","updated_at":"2026-07-05T07:17:49.985066+00:00"}