{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:YZUFNRDBJYMXBCETSAAQCG7425","short_pith_number":"pith:YZUFNRDB","schema_version":"1.0","canonical_sha256":"c66856c4614e197088939001011bfcd743bb3e269ae37efb208cf6249a8a85fc","source":{"kind":"arxiv","id":"2208.03550","version":1},"attestation_state":"computed","paper":{"title":"Frozen CLIP Models are Efficient Video Learners","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gerard de Melo, Hongsheng Li, Jifeng Dai, Peng Gao, Renrui Zhang, Shijie Geng, Xiaogang Wang, Yu Qiao, Ziyi Lin","submitted_at":"2022-08-06T17:38:25Z","abstract_excerpt":"Video recognition has been dominated by the end-to-end learning paradigm -- first initializing a video recognition model with weights of a pretrained image model and then conducting end-to-end training on videos. This enables the video network to benefit from the pretrained image model. However, this requires substantial computation and memory resources for finetuning on videos and the alternative of directly using pretrained image features without finetuning the image backbone leads to subpar results. Fortunately, recent advances in Contrastive Vision-Language Pre-training (CLIP) pave the way"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.03550","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-08-06T17:38:25Z","cross_cats_sorted":[],"title_canon_sha256":"148e5db14ae7b2a64c95935fb4650c790d0d4c2d90f1f8dfbe7476d812443aa1","abstract_canon_sha256":"96d17bf7cffa1a657ead4a6356210576552f861389f99585bec6ad01c0ca712b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:46:41.885971Z","signature_b64":"emvAYj0sC4vOwmXEpkO7JGjXJxkS4xK3Uy5ZyV4Z29+pAfb9iPPjFussvTnT6HuiOCeYK5UMGLM8KNKzZGSJBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c66856c4614e197088939001011bfcd743bb3e269ae37efb208cf6249a8a85fc","last_reissued_at":"2026-07-05T04:46:41.885494Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:46:41.885494Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Frozen CLIP Models are Efficient Video Learners","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gerard de Melo, Hongsheng Li, Jifeng Dai, Peng Gao, Renrui Zhang, Shijie Geng, Xiaogang Wang, Yu Qiao, Ziyi Lin","submitted_at":"2022-08-06T17:38:25Z","abstract_excerpt":"Video recognition has been dominated by the end-to-end learning paradigm -- first initializing a video recognition model with weights of a pretrained image model and then conducting end-to-end training on videos. This enables the video network to benefit from the pretrained image model. However, this requires substantial computation and memory resources for finetuning on videos and the alternative of directly using pretrained image features without finetuning the image backbone leads to subpar results. Fortunately, recent advances in Contrastive Vision-Language Pre-training (CLIP) pave the way"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.03550","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.03550/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.03550","created_at":"2026-07-05T04:46:41.885560+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.03550v1","created_at":"2026-07-05T04:46:41.885560+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.03550","created_at":"2026-07-05T04:46:41.885560+00:00"},{"alias_kind":"pith_short_12","alias_value":"YZUFNRDBJYMX","created_at":"2026-07-05T04:46:41.885560+00:00"},{"alias_kind":"pith_short_16","alias_value":"YZUFNRDBJYMXBCET","created_at":"2026-07-05T04:46:41.885560+00:00"},{"alias_kind":"pith_short_8","alias_value":"YZUFNRDB","created_at":"2026-07-05T04:46:41.885560+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09646","citing_title":"Do Video Foundation Models Understand Intuitive Physics? A Layerwise Probing Analysis","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2211.11018","citing_title":"MagicVideo: Efficient Video Generation With Latent Diffusion Models","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YZUFNRDBJYMXBCETSAAQCG7425","json":"https://pith.science/pith/YZUFNRDBJYMXBCETSAAQCG7425.json","graph_json":"https://pith.science/api/pith-number/YZUFNRDBJYMXBCETSAAQCG7425/graph.json","events_json":"https://pith.science/api/pith-number/YZUFNRDBJYMXBCETSAAQCG7425/events.json","paper":"https://pith.science/paper/YZUFNRDB"},"agent_actions":{"view_html":"https://pith.science/pith/YZUFNRDBJYMXBCETSAAQCG7425","download_json":"https://pith.science/pith/YZUFNRDBJYMXBCETSAAQCG7425.json","view_paper":"https://pith.science/paper/YZUFNRDB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.03550&json=true","fetch_graph":"https://pith.science/api/pith-number/YZUFNRDBJYMXBCETSAAQCG7425/graph.json","fetch_events":"https://pith.science/api/pith-number/YZUFNRDBJYMXBCETSAAQCG7425/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YZUFNRDBJYMXBCETSAAQCG7425/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YZUFNRDBJYMXBCETSAAQCG7425/action/storage_attestation","attest_author":"https://pith.science/pith/YZUFNRDBJYMXBCETSAAQCG7425/action/author_attestation","sign_citation":"https://pith.science/pith/YZUFNRDBJYMXBCETSAAQCG7425/action/citation_signature","submit_replication":"https://pith.science/pith/YZUFNRDBJYMXBCETSAAQCG7425/action/replication_record"}},"created_at":"2026-07-05T04:46:41.885560+00:00","updated_at":"2026-07-05T04:46:41.885560+00:00"}