{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VLD7OKTGLMIN6B5T4FU3QDXO6Q","short_pith_number":"pith:VLD7OKTG","schema_version":"1.0","canonical_sha256":"aac7f72a665b10df07b3e169b80eeef42850913e105e1f384ebe3f264a1e00b7","source":{"kind":"arxiv","id":"2308.03712","version":2},"attestation_state":"computed","paper":{"title":"Scaling may be all you need for achieving human-level object recognition capacity with human-like visual experience","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NE","q-bio.NC"],"primary_cat":"cs.CV","authors_text":"A. Emin Orhan","submitted_at":"2023-08-07T16:31:38Z","abstract_excerpt":"This paper asks whether current self-supervised learning methods, if sufficiently scaled up, would be able to reach human-level visual object recognition capabilities with the same type and amount of visual experience humans learn from. Previous work on this question only considered the scaling of data size. Here, we consider the simultaneous scaling of data size, model size, and image resolution. We perform a scaling experiment with vision transformers up to 633M parameters in size (ViT-H/14) trained with up to 5K hours of human-like video data (long, continuous, mostly egocentric videos) wit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.03712","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-08-07T16:31:38Z","cross_cats_sorted":["cs.LG","cs.NE","q-bio.NC"],"title_canon_sha256":"9cba57cca7e6e2220f43f6597b4ab63daf49aa373d330c5eb831f5ab899125cc","abstract_canon_sha256":"3002b90e024aa5deaf078c5ae9e25386fbbe92c241c9c95dc3fc5728e686ea17"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:39:55.198144Z","signature_b64":"S3C/IO5Pl9pPgsvmMEWC66ZSRM0dc9KTB0vMgcRzla7s4L88gw/hYJaVHWxmprrEnkLyyQLlcpkp8gX13fySDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aac7f72a665b10df07b3e169b80eeef42850913e105e1f384ebe3f264a1e00b7","last_reissued_at":"2026-07-05T06:39:55.197643Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:39:55.197643Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling may be all you need for achieving human-level object recognition capacity with human-like visual experience","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NE","q-bio.NC"],"primary_cat":"cs.CV","authors_text":"A. Emin Orhan","submitted_at":"2023-08-07T16:31:38Z","abstract_excerpt":"This paper asks whether current self-supervised learning methods, if sufficiently scaled up, would be able to reach human-level visual object recognition capabilities with the same type and amount of visual experience humans learn from. Previous work on this question only considered the scaling of data size. Here, we consider the simultaneous scaling of data size, model size, and image resolution. We perform a scaling experiment with vision transformers up to 633M parameters in size (ViT-H/14) trained with up to 5K hours of human-like video data (long, continuous, mostly egocentric videos) wit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.03712","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.03712/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.03712","created_at":"2026-07-05T06:39:55.197709+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.03712v2","created_at":"2026-07-05T06:39:55.197709+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.03712","created_at":"2026-07-05T06:39:55.197709+00:00"},{"alias_kind":"pith_short_12","alias_value":"VLD7OKTGLMIN","created_at":"2026-07-05T06:39:55.197709+00:00"},{"alias_kind":"pith_short_16","alias_value":"VLD7OKTGLMIN6B5T","created_at":"2026-07-05T06:39:55.197709+00:00"},{"alias_kind":"pith_short_8","alias_value":"VLD7OKTG","created_at":"2026-07-05T06:39:55.197709+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.02966","citing_title":"Human Gaze Boosts Object-Centered Representation Learning","ref_index":38,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VLD7OKTGLMIN6B5T4FU3QDXO6Q","json":"https://pith.science/pith/VLD7OKTGLMIN6B5T4FU3QDXO6Q.json","graph_json":"https://pith.science/api/pith-number/VLD7OKTGLMIN6B5T4FU3QDXO6Q/graph.json","events_json":"https://pith.science/api/pith-number/VLD7OKTGLMIN6B5T4FU3QDXO6Q/events.json","paper":"https://pith.science/paper/VLD7OKTG"},"agent_actions":{"view_html":"https://pith.science/pith/VLD7OKTGLMIN6B5T4FU3QDXO6Q","download_json":"https://pith.science/pith/VLD7OKTGLMIN6B5T4FU3QDXO6Q.json","view_paper":"https://pith.science/paper/VLD7OKTG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.03712&json=true","fetch_graph":"https://pith.science/api/pith-number/VLD7OKTGLMIN6B5T4FU3QDXO6Q/graph.json","fetch_events":"https://pith.science/api/pith-number/VLD7OKTGLMIN6B5T4FU3QDXO6Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VLD7OKTGLMIN6B5T4FU3QDXO6Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VLD7OKTGLMIN6B5T4FU3QDXO6Q/action/storage_attestation","attest_author":"https://pith.science/pith/VLD7OKTGLMIN6B5T4FU3QDXO6Q/action/author_attestation","sign_citation":"https://pith.science/pith/VLD7OKTGLMIN6B5T4FU3QDXO6Q/action/citation_signature","submit_replication":"https://pith.science/pith/VLD7OKTGLMIN6B5T4FU3QDXO6Q/action/replication_record"}},"created_at":"2026-07-05T06:39:55.197709+00:00","updated_at":"2026-07-05T06:39:55.197709+00:00"}