{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ROQMLI4TQFYQESSCAAOY7GIXZI","short_pith_number":"pith:ROQMLI4T","schema_version":"1.0","canonical_sha256":"8ba0c5a3938171024a42001d8f9917ca195b105a934ad805177880255e74d23c","source":{"kind":"arxiv","id":"2506.23283","version":1},"attestation_state":"computed","paper":{"title":"MoMa: Modulating Mamba for Adapting Image Foundation Models to Video Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chaofan Ma, Jiangchao Yao, Yanfeng Wang, Ya Zhang, Yuhuan Yang, Zhenjie Mao","submitted_at":"2025-06-29T15:14:55Z","abstract_excerpt":"Video understanding is a complex challenge that requires effective modeling of spatial-temporal dynamics. With the success of image foundation models (IFMs) in image understanding, recent approaches have explored parameter-efficient fine-tuning (PEFT) to adapt IFMs for video. However, most of these methods tend to process spatial and temporal information separately, which may fail to capture the full intricacy of video dynamics. In this paper, we propose MoMa, an efficient adapter framework that achieves full spatial-temporal modeling by integrating Mamba's selective state space modeling into "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.23283","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-29T15:14:55Z","cross_cats_sorted":[],"title_canon_sha256":"2f9934b42d3879ddb39fab5ec885b59930222f4c3bcc34a4db3bd4468af74acb","abstract_canon_sha256":"2b723443d6cff1b9c31f27ef5fef9667f5b2dcbf25d29b0f57ef8f98a6edfdd5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:29:19.197953Z","signature_b64":"QwqRC4pRZvo6iLEuh2NAQfRtTf2vyi6gplm2up6nh1GOdECSmZejDm3v+EI9WUvglzzjBKJ1Dfzw0h3jxO0LCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ba0c5a3938171024a42001d8f9917ca195b105a934ad805177880255e74d23c","last_reissued_at":"2026-07-05T11:29:19.197442Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:29:19.197442Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MoMa: Modulating Mamba for Adapting Image Foundation Models to Video Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chaofan Ma, Jiangchao Yao, Yanfeng Wang, Ya Zhang, Yuhuan Yang, Zhenjie Mao","submitted_at":"2025-06-29T15:14:55Z","abstract_excerpt":"Video understanding is a complex challenge that requires effective modeling of spatial-temporal dynamics. With the success of image foundation models (IFMs) in image understanding, recent approaches have explored parameter-efficient fine-tuning (PEFT) to adapt IFMs for video. However, most of these methods tend to process spatial and temporal information separately, which may fail to capture the full intricacy of video dynamics. In this paper, we propose MoMa, an efficient adapter framework that achieves full spatial-temporal modeling by integrating Mamba's selective state space modeling into "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.23283","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.23283/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.23283","created_at":"2026-07-05T11:29:19.197501+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.23283v1","created_at":"2026-07-05T11:29:19.197501+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.23283","created_at":"2026-07-05T11:29:19.197501+00:00"},{"alias_kind":"pith_short_12","alias_value":"ROQMLI4TQFYQ","created_at":"2026-07-05T11:29:19.197501+00:00"},{"alias_kind":"pith_short_16","alias_value":"ROQMLI4TQFYQESSC","created_at":"2026-07-05T11:29:19.197501+00:00"},{"alias_kind":"pith_short_8","alias_value":"ROQMLI4T","created_at":"2026-07-05T11:29:19.197501+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ROQMLI4TQFYQESSCAAOY7GIXZI","json":"https://pith.science/pith/ROQMLI4TQFYQESSCAAOY7GIXZI.json","graph_json":"https://pith.science/api/pith-number/ROQMLI4TQFYQESSCAAOY7GIXZI/graph.json","events_json":"https://pith.science/api/pith-number/ROQMLI4TQFYQESSCAAOY7GIXZI/events.json","paper":"https://pith.science/paper/ROQMLI4T"},"agent_actions":{"view_html":"https://pith.science/pith/ROQMLI4TQFYQESSCAAOY7GIXZI","download_json":"https://pith.science/pith/ROQMLI4TQFYQESSCAAOY7GIXZI.json","view_paper":"https://pith.science/paper/ROQMLI4T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.23283&json=true","fetch_graph":"https://pith.science/api/pith-number/ROQMLI4TQFYQESSCAAOY7GIXZI/graph.json","fetch_events":"https://pith.science/api/pith-number/ROQMLI4TQFYQESSCAAOY7GIXZI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ROQMLI4TQFYQESSCAAOY7GIXZI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ROQMLI4TQFYQESSCAAOY7GIXZI/action/storage_attestation","attest_author":"https://pith.science/pith/ROQMLI4TQFYQESSCAAOY7GIXZI/action/author_attestation","sign_citation":"https://pith.science/pith/ROQMLI4TQFYQESSCAAOY7GIXZI/action/citation_signature","submit_replication":"https://pith.science/pith/ROQMLI4TQFYQESSCAAOY7GIXZI/action/replication_record"}},"created_at":"2026-07-05T11:29:19.197501+00:00","updated_at":"2026-07-05T11:29:19.197501+00:00"}