{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:H57AYK3O64CAE4ZKJXPPFPCRPH","short_pith_number":"pith:H57AYK3O","schema_version":"1.0","canonical_sha256":"3f7e0c2b6ef70402732a4ddef2bc5179dca01a1c677e36c93521215ac2b8ff98","source":{"kind":"arxiv","id":"2411.19460","version":1},"attestation_state":"computed","paper":{"title":"Look Every Frame All at Once: Video-Ma$^2$mba for Efficient Long-form Video Understanding with Multi-Axis Gradient Checkpointing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Hosu Lee, Hyunjun Kim, Junho Kim, Yong Man Ro","submitted_at":"2024-11-29T04:12:13Z","abstract_excerpt":"With the growing scale and complexity of video data, efficiently processing long video sequences poses significant challenges due to the quadratic increase in memory and computational demands associated with existing transformer-based Large Multi-modal Models (LMMs). To address these issues, we introduce Video-Ma$^2$mba, a novel architecture that incorporates State Space Models (SSMs) within the Mamba-2 framework, replacing the attention mechanisms. This allows the LMMs to scale linearly in terms of time and memory requirements, making it feasible to handle long-duration video content. Further"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.19460","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-29T04:12:13Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"daef51a2397c6f5503f4a51890bc732fc16ca716c3f50ba18be65049545d2f10","abstract_canon_sha256":"d4344af2426a0882ee09ee60a0c28abe6f6de473be0cafac89fa89ef5e55a8e8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:42:02.819635Z","signature_b64":"/B65oJftwBGhWeEuEnJMAZGXd4h6zBK8ZbtyEvJRO1c0z16Fw2q3n3jEmPTwfH1+S4g/O6Hy6WqN3QyODH/6Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3f7e0c2b6ef70402732a4ddef2bc5179dca01a1c677e36c93521215ac2b8ff98","last_reissued_at":"2026-07-05T09:42:02.819133Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:42:02.819133Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Look Every Frame All at Once: Video-Ma$^2$mba for Efficient Long-form Video Understanding with Multi-Axis Gradient Checkpointing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Hosu Lee, Hyunjun Kim, Junho Kim, Yong Man Ro","submitted_at":"2024-11-29T04:12:13Z","abstract_excerpt":"With the growing scale and complexity of video data, efficiently processing long video sequences poses significant challenges due to the quadratic increase in memory and computational demands associated with existing transformer-based Large Multi-modal Models (LMMs). To address these issues, we introduce Video-Ma$^2$mba, a novel architecture that incorporates State Space Models (SSMs) within the Mamba-2 framework, replacing the attention mechanisms. This allows the LMMs to scale linearly in terms of time and memory requirements, making it feasible to handle long-duration video content. Further"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.19460","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.19460/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.19460","created_at":"2026-07-05T09:42:02.819198+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.19460v1","created_at":"2026-07-05T09:42:02.819198+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.19460","created_at":"2026-07-05T09:42:02.819198+00:00"},{"alias_kind":"pith_short_12","alias_value":"H57AYK3O64CA","created_at":"2026-07-05T09:42:02.819198+00:00"},{"alias_kind":"pith_short_16","alias_value":"H57AYK3O64CAE4ZK","created_at":"2026-07-05T09:42:02.819198+00:00"},{"alias_kind":"pith_short_8","alias_value":"H57AYK3O","created_at":"2026-07-05T09:42:02.819198+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H57AYK3O64CAE4ZKJXPPFPCRPH","json":"https://pith.science/pith/H57AYK3O64CAE4ZKJXPPFPCRPH.json","graph_json":"https://pith.science/api/pith-number/H57AYK3O64CAE4ZKJXPPFPCRPH/graph.json","events_json":"https://pith.science/api/pith-number/H57AYK3O64CAE4ZKJXPPFPCRPH/events.json","paper":"https://pith.science/paper/H57AYK3O"},"agent_actions":{"view_html":"https://pith.science/pith/H57AYK3O64CAE4ZKJXPPFPCRPH","download_json":"https://pith.science/pith/H57AYK3O64CAE4ZKJXPPFPCRPH.json","view_paper":"https://pith.science/paper/H57AYK3O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.19460&json=true","fetch_graph":"https://pith.science/api/pith-number/H57AYK3O64CAE4ZKJXPPFPCRPH/graph.json","fetch_events":"https://pith.science/api/pith-number/H57AYK3O64CAE4ZKJXPPFPCRPH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H57AYK3O64CAE4ZKJXPPFPCRPH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H57AYK3O64CAE4ZKJXPPFPCRPH/action/storage_attestation","attest_author":"https://pith.science/pith/H57AYK3O64CAE4ZKJXPPFPCRPH/action/author_attestation","sign_citation":"https://pith.science/pith/H57AYK3O64CAE4ZKJXPPFPCRPH/action/citation_signature","submit_replication":"https://pith.science/pith/H57AYK3O64CAE4ZKJXPPFPCRPH/action/replication_record"}},"created_at":"2026-07-05T09:42:02.819198+00:00","updated_at":"2026-07-05T09:42:02.819198+00:00"}