{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:JSCQRSGUZDFBSYLSDMGAYYOEQO","short_pith_number":"pith:JSCQRSGU","schema_version":"1.0","canonical_sha256":"4c8508c8d4c8ca1961721b0c0c61c483b08a93384532a164393fa6ca0abe97d9","source":{"kind":"arxiv","id":"2205.01721","version":2},"attestation_state":"computed","paper":{"title":"In Defense of Image Pre-Training for Spatiotemporal Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Chen Wei, Cihang Xie, Huiyu Wang, Jieru Mei, Xianhang Li, Yuyin Zhou","submitted_at":"2022-05-03T18:45:44Z","abstract_excerpt":"Image pre-training, the current de-facto paradigm for a wide range of visual tasks, is generally less favored in the field of video recognition. By contrast, a common strategy is to directly train with spatiotemporal convolutional neural networks (CNNs) from scratch. Nonetheless, interestingly, by taking a closer look at these from-scratch learned CNNs, we note there exist certain 3D kernels that exhibit much stronger appearance modeling ability than others, arguably suggesting appearance information is already well disentangled in learning. Inspired by this observation, we hypothesize that th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.01721","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-05-03T18:45:44Z","cross_cats_sorted":[],"title_canon_sha256":"c36bc34d85aeeaa73d6e74cd1a59dc854eed04b73e32bf7e5a289776ebdd913c","abstract_canon_sha256":"cb341ba8182215ac7f8c56941cfeca369762a1aff8e3ab80db7d11b33d1e1b6a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:45:20.601241Z","signature_b64":"T0XF1ARrXtQYYX8CM5kMWUc7bhxXWK4uXl7Wfk9MTB39oK2T8SZX+2tbRutSMij+hcdhS0QmpwKlS3QxqCx3BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c8508c8d4c8ca1961721b0c0c61c483b08a93384532a164393fa6ca0abe97d9","last_reissued_at":"2026-07-05T04:45:20.600725Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:45:20.600725Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"In Defense of Image Pre-Training for Spatiotemporal Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Chen Wei, Cihang Xie, Huiyu Wang, Jieru Mei, Xianhang Li, Yuyin Zhou","submitted_at":"2022-05-03T18:45:44Z","abstract_excerpt":"Image pre-training, the current de-facto paradigm for a wide range of visual tasks, is generally less favored in the field of video recognition. By contrast, a common strategy is to directly train with spatiotemporal convolutional neural networks (CNNs) from scratch. Nonetheless, interestingly, by taking a closer look at these from-scratch learned CNNs, we note there exist certain 3D kernels that exhibit much stronger appearance modeling ability than others, arguably suggesting appearance information is already well disentangled in learning. Inspired by this observation, we hypothesize that th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.01721","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.01721/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.01721","created_at":"2026-07-05T04:45:20.600786+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.01721v2","created_at":"2026-07-05T04:45:20.600786+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.01721","created_at":"2026-07-05T04:45:20.600786+00:00"},{"alias_kind":"pith_short_12","alias_value":"JSCQRSGUZDFB","created_at":"2026-07-05T04:45:20.600786+00:00"},{"alias_kind":"pith_short_16","alias_value":"JSCQRSGUZDFBSYLS","created_at":"2026-07-05T04:45:20.600786+00:00"},{"alias_kind":"pith_short_8","alias_value":"JSCQRSGU","created_at":"2026-07-05T04:45:20.600786+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JSCQRSGUZDFBSYLSDMGAYYOEQO","json":"https://pith.science/pith/JSCQRSGUZDFBSYLSDMGAYYOEQO.json","graph_json":"https://pith.science/api/pith-number/JSCQRSGUZDFBSYLSDMGAYYOEQO/graph.json","events_json":"https://pith.science/api/pith-number/JSCQRSGUZDFBSYLSDMGAYYOEQO/events.json","paper":"https://pith.science/paper/JSCQRSGU"},"agent_actions":{"view_html":"https://pith.science/pith/JSCQRSGUZDFBSYLSDMGAYYOEQO","download_json":"https://pith.science/pith/JSCQRSGUZDFBSYLSDMGAYYOEQO.json","view_paper":"https://pith.science/paper/JSCQRSGU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.01721&json=true","fetch_graph":"https://pith.science/api/pith-number/JSCQRSGUZDFBSYLSDMGAYYOEQO/graph.json","fetch_events":"https://pith.science/api/pith-number/JSCQRSGUZDFBSYLSDMGAYYOEQO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JSCQRSGUZDFBSYLSDMGAYYOEQO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JSCQRSGUZDFBSYLSDMGAYYOEQO/action/storage_attestation","attest_author":"https://pith.science/pith/JSCQRSGUZDFBSYLSDMGAYYOEQO/action/author_attestation","sign_citation":"https://pith.science/pith/JSCQRSGUZDFBSYLSDMGAYYOEQO/action/citation_signature","submit_replication":"https://pith.science/pith/JSCQRSGUZDFBSYLSDMGAYYOEQO/action/replication_record"}},"created_at":"2026-07-05T04:45:20.600786+00:00","updated_at":"2026-07-05T04:45:20.600786+00:00"}