{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RCEMPZ3C653PEBYFCCTKY426HD","short_pith_number":"pith:RCEMPZ3C","schema_version":"1.0","canonical_sha256":"8888c7e762f776f2070510a6ac735e38ee74cee089d7f17656f1cf9ff02031f1","source":{"kind":"arxiv","id":"2401.11649","version":1},"attestation_state":"computed","paper":{"title":"M2-CLIP: A Multimodal, Multi-task Adapting Framework for Video Action Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Boyuan Jiang, Guang Dai, Jianbiao Mei, Jiazheng Xing, Jingdong Wang, Jun Chen, Mengmeng Wang, Xingxing Zuo, Yong Liu","submitted_at":"2024-01-22T02:03:31Z","abstract_excerpt":"Recently, the rise of large-scale vision-language pretrained models like CLIP, coupled with the technology of Parameter-Efficient FineTuning (PEFT), has captured substantial attraction in video action recognition. Nevertheless, prevailing approaches tend to prioritize strong supervised performance at the expense of compromising the models' generalization capabilities during transfer. In this paper, we introduce a novel Multimodal, Multi-task CLIP adapting framework named \\name to address these challenges, preserving both high supervised performance and robust transferability. Firstly, to enhan"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.11649","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-22T02:03:31Z","cross_cats_sorted":[],"title_canon_sha256":"0bb00ceff3a99386e11aa5c35fe06c58a4e50deab1bbbca951bc32a4db6d401e","abstract_canon_sha256":"934bec99a547ca3428a472c838a808cfaf9a4c7916ed538741e33f839df334e7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:36:07.884130Z","signature_b64":"VD8/yNr8pxnmWISjDGKZRbpS+k0nb4DMHBXmr4DaX/TN3xUV8OTlRb3eftr0S2pRLZ7U0kNkfe13VX6pgCZ2DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8888c7e762f776f2070510a6ac735e38ee74cee089d7f17656f1cf9ff02031f1","last_reissued_at":"2026-07-05T07:36:07.883665Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:36:07.883665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"M2-CLIP: A Multimodal, Multi-task Adapting Framework for Video Action Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Boyuan Jiang, Guang Dai, Jianbiao Mei, Jiazheng Xing, Jingdong Wang, Jun Chen, Mengmeng Wang, Xingxing Zuo, Yong Liu","submitted_at":"2024-01-22T02:03:31Z","abstract_excerpt":"Recently, the rise of large-scale vision-language pretrained models like CLIP, coupled with the technology of Parameter-Efficient FineTuning (PEFT), has captured substantial attraction in video action recognition. Nevertheless, prevailing approaches tend to prioritize strong supervised performance at the expense of compromising the models' generalization capabilities during transfer. In this paper, we introduce a novel Multimodal, Multi-task CLIP adapting framework named \\name to address these challenges, preserving both high supervised performance and robust transferability. Firstly, to enhan"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.11649","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.11649/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.11649","created_at":"2026-07-05T07:36:07.883716+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.11649v1","created_at":"2026-07-05T07:36:07.883716+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.11649","created_at":"2026-07-05T07:36:07.883716+00:00"},{"alias_kind":"pith_short_12","alias_value":"RCEMPZ3C653P","created_at":"2026-07-05T07:36:07.883716+00:00"},{"alias_kind":"pith_short_16","alias_value":"RCEMPZ3C653PEBYF","created_at":"2026-07-05T07:36:07.883716+00:00"},{"alias_kind":"pith_short_8","alias_value":"RCEMPZ3C","created_at":"2026-07-05T07:36:07.883716+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.20760","citing_title":"Exploring High-Order Self-Similarity for Video Understanding","ref_index":84,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RCEMPZ3C653PEBYFCCTKY426HD","json":"https://pith.science/pith/RCEMPZ3C653PEBYFCCTKY426HD.json","graph_json":"https://pith.science/api/pith-number/RCEMPZ3C653PEBYFCCTKY426HD/graph.json","events_json":"https://pith.science/api/pith-number/RCEMPZ3C653PEBYFCCTKY426HD/events.json","paper":"https://pith.science/paper/RCEMPZ3C"},"agent_actions":{"view_html":"https://pith.science/pith/RCEMPZ3C653PEBYFCCTKY426HD","download_json":"https://pith.science/pith/RCEMPZ3C653PEBYFCCTKY426HD.json","view_paper":"https://pith.science/paper/RCEMPZ3C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.11649&json=true","fetch_graph":"https://pith.science/api/pith-number/RCEMPZ3C653PEBYFCCTKY426HD/graph.json","fetch_events":"https://pith.science/api/pith-number/RCEMPZ3C653PEBYFCCTKY426HD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RCEMPZ3C653PEBYFCCTKY426HD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RCEMPZ3C653PEBYFCCTKY426HD/action/storage_attestation","attest_author":"https://pith.science/pith/RCEMPZ3C653PEBYFCCTKY426HD/action/author_attestation","sign_citation":"https://pith.science/pith/RCEMPZ3C653PEBYFCCTKY426HD/action/citation_signature","submit_replication":"https://pith.science/pith/RCEMPZ3C653PEBYFCCTKY426HD/action/replication_record"}},"created_at":"2026-07-05T07:36:07.883716+00:00","updated_at":"2026-07-05T07:36:07.883716+00:00"}