{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NDEMKEKEPAUYPIMLQMMVWLGFZS","short_pith_number":"pith:NDEMKEKE","schema_version":"1.0","canonical_sha256":"68c8c51144782987a18b83195b2cc5cc87da1fbdb737f2dd011b0a9dc03191aa","source":{"kind":"arxiv","id":"2503.13707","version":1},"attestation_state":"computed","paper":{"title":"Long-VMNet: Accelerating Long-Form Video Understanding via Fixed Memory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Asim Kadav, Saket Gurukar","submitted_at":"2025-03-17T20:25:41Z","abstract_excerpt":"Long-form video understanding is essential for various applications such as video retrieval, summarizing, and question answering. Yet, traditional approaches demand substantial computing power and are often bottlenecked by GPU memory. To tackle this issue, we present Long-Video Memory Network, Long-VMNet, a novel video understanding method that employs a fixed-size memory representation to store discriminative patches sampled from the input video. Long-VMNet achieves improved efficiency by leveraging a neural sampler that identifies discriminative tokens. Additionally, Long-VMNet only needs on"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.13707","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-17T20:25:41Z","cross_cats_sorted":[],"title_canon_sha256":"0c4102b46880a684eb54efdb9ba3552912d317312870da47a8be1a0bd0addc38","abstract_canon_sha256":"70fa7113561422b9bd6e23163548694cd4cc4baa7f767b01b751737a73be4400"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:33:16.314521Z","signature_b64":"/oU+uYhyr8EE1USm5CYK+EarzBTMupBR2EhCaHXxoTSEm6+fZS7InsdtXZ50FwAZlgoRTG9WkXigElQ6cRKKDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"68c8c51144782987a18b83195b2cc5cc87da1fbdb737f2dd011b0a9dc03191aa","last_reissued_at":"2026-07-05T10:33:16.314010Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:33:16.314010Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Long-VMNet: Accelerating Long-Form Video Understanding via Fixed Memory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Asim Kadav, Saket Gurukar","submitted_at":"2025-03-17T20:25:41Z","abstract_excerpt":"Long-form video understanding is essential for various applications such as video retrieval, summarizing, and question answering. Yet, traditional approaches demand substantial computing power and are often bottlenecked by GPU memory. To tackle this issue, we present Long-Video Memory Network, Long-VMNet, a novel video understanding method that employs a fixed-size memory representation to store discriminative patches sampled from the input video. Long-VMNet achieves improved efficiency by leveraging a neural sampler that identifies discriminative tokens. Additionally, Long-VMNet only needs on"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.13707","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.13707/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.13707","created_at":"2026-07-05T10:33:16.314078+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.13707v1","created_at":"2026-07-05T10:33:16.314078+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.13707","created_at":"2026-07-05T10:33:16.314078+00:00"},{"alias_kind":"pith_short_12","alias_value":"NDEMKEKEPAUY","created_at":"2026-07-05T10:33:16.314078+00:00"},{"alias_kind":"pith_short_16","alias_value":"NDEMKEKEPAUYPIML","created_at":"2026-07-05T10:33:16.314078+00:00"},{"alias_kind":"pith_short_8","alias_value":"NDEMKEKE","created_at":"2026-07-05T10:33:16.314078+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19849","citing_title":"ViCoStream: Streaming VideoLLMs Can Run Beyond 100 FPS with Stage-Wise Coordinated Inference","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13080","citing_title":"Learning to See What You Need: Gaze Attention for Multimodal Large Language Models","ref_index":137,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NDEMKEKEPAUYPIMLQMMVWLGFZS","json":"https://pith.science/pith/NDEMKEKEPAUYPIMLQMMVWLGFZS.json","graph_json":"https://pith.science/api/pith-number/NDEMKEKEPAUYPIMLQMMVWLGFZS/graph.json","events_json":"https://pith.science/api/pith-number/NDEMKEKEPAUYPIMLQMMVWLGFZS/events.json","paper":"https://pith.science/paper/NDEMKEKE"},"agent_actions":{"view_html":"https://pith.science/pith/NDEMKEKEPAUYPIMLQMMVWLGFZS","download_json":"https://pith.science/pith/NDEMKEKEPAUYPIMLQMMVWLGFZS.json","view_paper":"https://pith.science/paper/NDEMKEKE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.13707&json=true","fetch_graph":"https://pith.science/api/pith-number/NDEMKEKEPAUYPIMLQMMVWLGFZS/graph.json","fetch_events":"https://pith.science/api/pith-number/NDEMKEKEPAUYPIMLQMMVWLGFZS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NDEMKEKEPAUYPIMLQMMVWLGFZS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NDEMKEKEPAUYPIMLQMMVWLGFZS/action/storage_attestation","attest_author":"https://pith.science/pith/NDEMKEKEPAUYPIMLQMMVWLGFZS/action/author_attestation","sign_citation":"https://pith.science/pith/NDEMKEKEPAUYPIMLQMMVWLGFZS/action/citation_signature","submit_replication":"https://pith.science/pith/NDEMKEKEPAUYPIMLQMMVWLGFZS/action/replication_record"}},"created_at":"2026-07-05T10:33:16.314078+00:00","updated_at":"2026-07-05T10:33:16.314078+00:00"}