{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:62AEYVZ3VUMZPNK2J4H2P5MQ3V","short_pith_number":"pith:62AEYVZ3","schema_version":"1.0","canonical_sha256":"f6804c573bad1997b55a4f0fa7f590dd5c5fe1887548ca0c104cb86e074167c3","source":{"kind":"arxiv","id":"2401.08541","version":1},"attestation_state":"computed","paper":{"title":"Scalable Pre-training of Large Autoregressive Image Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alaaeldin El-Nouby, Alexander Toshev, Armand Joulin, Joshua M Susskind, Michal Klein, Miguel Angel Bautista, Shuangfei Zhai, Vaishaal Shankar","submitted_at":"2024-01-16T18:03:37Z","abstract_excerpt":"This paper introduces AIM, a collection of vision models pre-trained with an autoregressive objective. These models are inspired by their textual counterparts, i.e., Large Language Models (LLMs), and exhibit similar scaling properties. Specifically, we highlight two key findings: (1) the performance of the visual features scale with both the model capacity and the quantity of data, (2) the value of the objective function correlates with the performance of the model on downstream tasks. We illustrate the practical implication of these findings by pre-training a 7 billion parameter AIM on 2 bill"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.08541","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-01-16T18:03:37Z","cross_cats_sorted":[],"title_canon_sha256":"f40483fd5400365b066c693ec2b1875a04bfc60ac4ded4a82f5f35170ac827d1","abstract_canon_sha256":"5e5372ee8c24224cdcb4c3b19f5991f7f6875e80088707011d270a2458bb6458"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:34:13.298753Z","signature_b64":"a/PO/lenAp8l5MqCLLMPlb2e6hY/w17WaQyC3wsBjmc2/YQaEISrnJp1jAWOZAEOvvysCFk4yPKPCndoFzCgAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f6804c573bad1997b55a4f0fa7f590dd5c5fe1887548ca0c104cb86e074167c3","last_reissued_at":"2026-07-05T07:34:13.298282Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:34:13.298282Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scalable Pre-training of Large Autoregressive Image Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alaaeldin El-Nouby, Alexander Toshev, Armand Joulin, Joshua M Susskind, Michal Klein, Miguel Angel Bautista, Shuangfei Zhai, Vaishaal Shankar","submitted_at":"2024-01-16T18:03:37Z","abstract_excerpt":"This paper introduces AIM, a collection of vision models pre-trained with an autoregressive objective. These models are inspired by their textual counterparts, i.e., Large Language Models (LLMs), and exhibit similar scaling properties. Specifically, we highlight two key findings: (1) the performance of the visual features scale with both the model capacity and the quantity of data, (2) the value of the objective function correlates with the performance of the model on downstream tasks. We illustrate the practical implication of these findings by pre-training a 7 billion parameter AIM on 2 bill"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.08541","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.08541/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.08541","created_at":"2026-07-05T07:34:13.298340+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.08541v1","created_at":"2026-07-05T07:34:13.298340+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.08541","created_at":"2026-07-05T07:34:13.298340+00:00"},{"alias_kind":"pith_short_12","alias_value":"62AEYVZ3VUMZ","created_at":"2026-07-05T07:34:13.298340+00:00"},{"alias_kind":"pith_short_16","alias_value":"62AEYVZ3VUMZPNK2","created_at":"2026-07-05T07:34:13.298340+00:00"},{"alias_kind":"pith_short_8","alias_value":"62AEYVZ3","created_at":"2026-07-05T07:34:13.298340+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26794","citing_title":"ReasonCLIP-58M: Visually Grounded Commonsense Reasoning Supervision for CLIP","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23992","citing_title":"A World Model of Radiologist Reading for Medical Image Representation Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25610","citing_title":"The Galaxy's Guide to the Tokenizer: A Benchmark for Scientific Foundation Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23033","citing_title":"Uncovering the Latent Potential of Deep Intermediate Representations","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17472","citing_title":"Weighted Reverse Convolution for Feature Upsampling","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16384","citing_title":"Mutual Enhancement Between Global Tokens and Patch Tokens: From Theory to Practice","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17472","citing_title":"Weighted Reverse Convolution for Feature Upsampling","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2403.09611","citing_title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08298","citing_title":"What Cohort INRs Encode and Where to Freeze Them","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01844","citing_title":"SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17961","citing_title":"DifFoundMAD: Foundation Models meet Differential Morphing Attack Detection","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/62AEYVZ3VUMZPNK2J4H2P5MQ3V","json":"https://pith.science/pith/62AEYVZ3VUMZPNK2J4H2P5MQ3V.json","graph_json":"https://pith.science/api/pith-number/62AEYVZ3VUMZPNK2J4H2P5MQ3V/graph.json","events_json":"https://pith.science/api/pith-number/62AEYVZ3VUMZPNK2J4H2P5MQ3V/events.json","paper":"https://pith.science/paper/62AEYVZ3"},"agent_actions":{"view_html":"https://pith.science/pith/62AEYVZ3VUMZPNK2J4H2P5MQ3V","download_json":"https://pith.science/pith/62AEYVZ3VUMZPNK2J4H2P5MQ3V.json","view_paper":"https://pith.science/paper/62AEYVZ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.08541&json=true","fetch_graph":"https://pith.science/api/pith-number/62AEYVZ3VUMZPNK2J4H2P5MQ3V/graph.json","fetch_events":"https://pith.science/api/pith-number/62AEYVZ3VUMZPNK2J4H2P5MQ3V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/62AEYVZ3VUMZPNK2J4H2P5MQ3V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/62AEYVZ3VUMZPNK2J4H2P5MQ3V/action/storage_attestation","attest_author":"https://pith.science/pith/62AEYVZ3VUMZPNK2J4H2P5MQ3V/action/author_attestation","sign_citation":"https://pith.science/pith/62AEYVZ3VUMZPNK2J4H2P5MQ3V/action/citation_signature","submit_replication":"https://pith.science/pith/62AEYVZ3VUMZPNK2J4H2P5MQ3V/action/replication_record"}},"created_at":"2026-07-05T07:34:13.298340+00:00","updated_at":"2026-07-05T07:34:13.298340+00:00"}