{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CFQURI7LSGTWPBN3ZRWEXD33IS","short_pith_number":"pith:CFQURI7L","schema_version":"1.0","canonical_sha256":"116148a3eb91a76785bbcc6c4b8f7b448f181269d4992fa9bf6dd479f8b4b96e","source":{"kind":"arxiv","id":"2404.03384","version":3},"attestation_state":"computed","paper":{"title":"LongVLM: Efficient Long Video Understanding via Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohan Zhuang, Haoyu He, Mingfei Han, Xiaojun Chang, Yuetian Weng","submitted_at":"2024-04-04T11:33:29Z","abstract_excerpt":"Empowered by Large Language Models (LLMs), recent advancements in Video-based LLMs (VideoLLMs) have driven progress in various video understanding tasks. These models encode video representations through pooling or query aggregation over a vast number of visual tokens, making computational and memory costs affordable. Despite successfully providing an overall comprehension of video content, existing VideoLLMs still face challenges in achieving detailed understanding due to overlooking local information in long-term videos. To tackle this challenge, we introduce LongVLM, a simple yet powerful V"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.03384","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-04-04T11:33:29Z","cross_cats_sorted":[],"title_canon_sha256":"d81cf37d8d3bcb2b72b4c0f9bd8f6039c5b59c66e5104bd7066c23c715650158","abstract_canon_sha256":"2e3b5b1373f94dd00ad1c315dfcfe65c29054cfc07df1effe5ddcb0b159fea95"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:46:23.724536Z","signature_b64":"kgZYxbUdRWgcUl4mg+xJHKtcs1BK6vMP2kiw+eYA4z/vvLiOys7IYb7I96r07iOI7t+6q2wvkKEyxQTtgn9QDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"116148a3eb91a76785bbcc6c4b8f7b448f181269d4992fa9bf6dd479f8b4b96e","last_reissued_at":"2026-07-05T08:46:23.724055Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:46:23.724055Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LongVLM: Efficient Long Video Understanding via Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohan Zhuang, Haoyu He, Mingfei Han, Xiaojun Chang, Yuetian Weng","submitted_at":"2024-04-04T11:33:29Z","abstract_excerpt":"Empowered by Large Language Models (LLMs), recent advancements in Video-based LLMs (VideoLLMs) have driven progress in various video understanding tasks. These models encode video representations through pooling or query aggregation over a vast number of visual tokens, making computational and memory costs affordable. Despite successfully providing an overall comprehension of video content, existing VideoLLMs still face challenges in achieving detailed understanding due to overlooking local information in long-term videos. To tackle this challenge, we introduce LongVLM, a simple yet powerful V"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.03384","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.03384/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.03384","created_at":"2026-07-05T08:46:23.724111+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.03384v3","created_at":"2026-07-05T08:46:23.724111+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.03384","created_at":"2026-07-05T08:46:23.724111+00:00"},{"alias_kind":"pith_short_12","alias_value":"CFQURI7LSGTW","created_at":"2026-07-05T08:46:23.724111+00:00"},{"alias_kind":"pith_short_16","alias_value":"CFQURI7LSGTWPBN3","created_at":"2026-07-05T08:46:23.724111+00:00"},{"alias_kind":"pith_short_8","alias_value":"CFQURI7L","created_at":"2026-07-05T08:46:23.724111+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2412.02930","citing_title":"TemporalVLM: Video LLMs for Temporal Reasoning in Long Videos","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2407.03320","citing_title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","ref_index":155,"is_internal_anchor":false},{"citing_arxiv_id":"2408.10188","citing_title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CFQURI7LSGTWPBN3ZRWEXD33IS","json":"https://pith.science/pith/CFQURI7LSGTWPBN3ZRWEXD33IS.json","graph_json":"https://pith.science/api/pith-number/CFQURI7LSGTWPBN3ZRWEXD33IS/graph.json","events_json":"https://pith.science/api/pith-number/CFQURI7LSGTWPBN3ZRWEXD33IS/events.json","paper":"https://pith.science/paper/CFQURI7L"},"agent_actions":{"view_html":"https://pith.science/pith/CFQURI7LSGTWPBN3ZRWEXD33IS","download_json":"https://pith.science/pith/CFQURI7LSGTWPBN3ZRWEXD33IS.json","view_paper":"https://pith.science/paper/CFQURI7L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.03384&json=true","fetch_graph":"https://pith.science/api/pith-number/CFQURI7LSGTWPBN3ZRWEXD33IS/graph.json","fetch_events":"https://pith.science/api/pith-number/CFQURI7LSGTWPBN3ZRWEXD33IS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CFQURI7LSGTWPBN3ZRWEXD33IS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CFQURI7LSGTWPBN3ZRWEXD33IS/action/storage_attestation","attest_author":"https://pith.science/pith/CFQURI7LSGTWPBN3ZRWEXD33IS/action/author_attestation","sign_citation":"https://pith.science/pith/CFQURI7LSGTWPBN3ZRWEXD33IS/action/citation_signature","submit_replication":"https://pith.science/pith/CFQURI7LSGTWPBN3ZRWEXD33IS/action/replication_record"}},"created_at":"2026-07-05T08:46:23.724111+00:00","updated_at":"2026-07-05T08:46:23.724111+00:00"}