{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UCFCFNMTL63PSSLFZA5SZOFGVX","short_pith_number":"pith:UCFCFNMT","schema_version":"1.0","canonical_sha256":"a08a22b5935fb6f94965c83b2cb8a6adc9b8b023ba693c545febaba406ab67ff","source":{"kind":"arxiv","id":"2303.13496","version":3},"attestation_state":"computed","paper":{"title":"The effectiveness of MAE pre-pretraining for billion-scale pretraining","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aaron Adcock, Armand Joulin, Christoph Feichtenhofer, Haoqi Fan, Ishan Misra, Kalyan Vasudev Alwala, Mannat Singh, Piotr Doll\\'ar, Quentin Duval, Rohit Girdhar, Ross Girshick, Vaibhav Aggarwal","submitted_at":"2023-03-23T17:56:12Z","abstract_excerpt":"This paper revisits the standard pretrain-then-finetune paradigm used in computer vision for visual recognition tasks. Typically, state-of-the-art foundation models are pretrained using large scale (weakly) supervised datasets with billions of images. We introduce an additional pre-pretraining stage that is simple and uses the self-supervised MAE technique to initialize the model. While MAE has only been shown to scale with the size of models, we find that it scales with the size of the training dataset as well. Thus, our MAE-based pre-pretraining scales with both model and data size making it"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.13496","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-03-23T17:56:12Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"1d4aedd7b87c5f47f70ca41afec4db4c3318d55c52b7632967709250744fd450","abstract_canon_sha256":"f95d73627fec31bb3ee4d0f6732d25b948cd0c6710d023fec311687d5187517f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:37:10.940217Z","signature_b64":"Gq32gkyU7DdSi4WMN3LjjKAfKnb8D4DehbffwLA/oMtZR5BhiR8OOZQ22vByRxzmdsxEmJxXfSsuMHHVicXrCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a08a22b5935fb6f94965c83b2cb8a6adc9b8b023ba693c545febaba406ab67ff","last_reissued_at":"2026-07-05T07:37:10.939700Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:37:10.939700Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The effectiveness of MAE pre-pretraining for billion-scale pretraining","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aaron Adcock, Armand Joulin, Christoph Feichtenhofer, Haoqi Fan, Ishan Misra, Kalyan Vasudev Alwala, Mannat Singh, Piotr Doll\\'ar, Quentin Duval, Rohit Girdhar, Ross Girshick, Vaibhav Aggarwal","submitted_at":"2023-03-23T17:56:12Z","abstract_excerpt":"This paper revisits the standard pretrain-then-finetune paradigm used in computer vision for visual recognition tasks. Typically, state-of-the-art foundation models are pretrained using large scale (weakly) supervised datasets with billions of images. We introduce an additional pre-pretraining stage that is simple and uses the self-supervised MAE technique to initialize the model. While MAE has only been shown to scale with the size of models, we find that it scales with the size of the training dataset as well. Thus, our MAE-based pre-pretraining scales with both model and data size making it"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.13496","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.13496/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.13496","created_at":"2026-07-05T07:37:10.939771+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.13496v3","created_at":"2026-07-05T07:37:10.939771+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.13496","created_at":"2026-07-05T07:37:10.939771+00:00"},{"alias_kind":"pith_short_12","alias_value":"UCFCFNMTL63P","created_at":"2026-07-05T07:37:10.939771+00:00"},{"alias_kind":"pith_short_16","alias_value":"UCFCFNMTL63PSSLF","created_at":"2026-07-05T07:37:10.939771+00:00"},{"alias_kind":"pith_short_8","alias_value":"UCFCFNMT","created_at":"2026-07-05T07:37:10.939771+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2312.14238","citing_title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","ref_index":129,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UCFCFNMTL63PSSLFZA5SZOFGVX","json":"https://pith.science/pith/UCFCFNMTL63PSSLFZA5SZOFGVX.json","graph_json":"https://pith.science/api/pith-number/UCFCFNMTL63PSSLFZA5SZOFGVX/graph.json","events_json":"https://pith.science/api/pith-number/UCFCFNMTL63PSSLFZA5SZOFGVX/events.json","paper":"https://pith.science/paper/UCFCFNMT"},"agent_actions":{"view_html":"https://pith.science/pith/UCFCFNMTL63PSSLFZA5SZOFGVX","download_json":"https://pith.science/pith/UCFCFNMTL63PSSLFZA5SZOFGVX.json","view_paper":"https://pith.science/paper/UCFCFNMT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.13496&json=true","fetch_graph":"https://pith.science/api/pith-number/UCFCFNMTL63PSSLFZA5SZOFGVX/graph.json","fetch_events":"https://pith.science/api/pith-number/UCFCFNMTL63PSSLFZA5SZOFGVX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UCFCFNMTL63PSSLFZA5SZOFGVX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UCFCFNMTL63PSSLFZA5SZOFGVX/action/storage_attestation","attest_author":"https://pith.science/pith/UCFCFNMTL63PSSLFZA5SZOFGVX/action/author_attestation","sign_citation":"https://pith.science/pith/UCFCFNMTL63PSSLFZA5SZOFGVX/action/citation_signature","submit_replication":"https://pith.science/pith/UCFCFNMTL63PSSLFZA5SZOFGVX/action/replication_record"}},"created_at":"2026-07-05T07:37:10.939771+00:00","updated_at":"2026-07-05T07:37:10.939771+00:00"}