{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7DS5SMLSJH2YKVQN4GQRWGGH6W","short_pith_number":"pith:7DS5SMLS","schema_version":"1.0","canonical_sha256":"f8e5d9317249f585560de1a11b18c7f589f34284fb971513d3bd51f61c8d3cdc","source":{"kind":"arxiv","id":"2403.13043","version":2},"attestation_state":"computed","paper":{"title":"When Do We Not Need Larger Vision Models?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baifeng Shi, Maolin Mao, Trevor Darrell, Xin Wang, Ziyang Wu","submitted_at":"2024-03-19T17:58:39Z","abstract_excerpt":"Scaling up the size of vision models has been the de facto standard to obtain more powerful visual representations. In this work, we discuss the point beyond which larger vision models are not necessary. First, we demonstrate the power of Scaling on Scales (S$^2$), whereby a pre-trained and frozen smaller vision model (e.g., ViT-B or ViT-L), run over multiple image scales, can outperform larger models (e.g., ViT-H or ViT-G) on classification, segmentation, depth estimation, Multimodal LLM (MLLM) benchmarks, and robotic manipulation. Notably, S$^2$ achieves state-of-the-art performance in detai"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.13043","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-19T17:58:39Z","cross_cats_sorted":[],"title_canon_sha256":"1dce7147772e326f18bca9435b1a390da3ff3b070532883c251c7bd9ffc7f4d6","abstract_canon_sha256":"56e0bbc74ef3ba1286cba10bf8118543d3fa04f91e423f1e86ff2aca7d86045b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:26.459455Z","signature_b64":"2OU+lBBRClzwDI6+TDioVnezm4G+axN4xJgXmOvdRhHCfkqrZ9hJ9K33BZaNZmoLbNqhf7mc34Lz9xAknt96Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f8e5d9317249f585560de1a11b18c7f589f34284fb971513d3bd51f61c8d3cdc","last_reissued_at":"2026-07-05T08:45:26.459005Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:26.459005Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Do We Not Need Larger Vision Models?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baifeng Shi, Maolin Mao, Trevor Darrell, Xin Wang, Ziyang Wu","submitted_at":"2024-03-19T17:58:39Z","abstract_excerpt":"Scaling up the size of vision models has been the de facto standard to obtain more powerful visual representations. In this work, we discuss the point beyond which larger vision models are not necessary. First, we demonstrate the power of Scaling on Scales (S$^2$), whereby a pre-trained and frozen smaller vision model (e.g., ViT-B or ViT-L), run over multiple image scales, can outperform larger models (e.g., ViT-H or ViT-G) on classification, segmentation, depth estimation, Multimodal LLM (MLLM) benchmarks, and robotic manipulation. Notably, S$^2$ achieves state-of-the-art performance in detai"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.13043","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.13043/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.13043","created_at":"2026-07-05T08:45:26.459062+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.13043v2","created_at":"2026-07-05T08:45:26.459062+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.13043","created_at":"2026-07-05T08:45:26.459062+00:00"},{"alias_kind":"pith_short_12","alias_value":"7DS5SMLSJH2Y","created_at":"2026-07-05T08:45:26.459062+00:00"},{"alias_kind":"pith_short_16","alias_value":"7DS5SMLSJH2YKVQN","created_at":"2026-07-05T08:45:26.459062+00:00"},{"alias_kind":"pith_short_8","alias_value":"7DS5SMLS","created_at":"2026-07-05T08:45:26.459062+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.18543","citing_title":"ClawEnvKit: Automatic Environment Generation for Claw-Like Agents","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03148","citing_title":"$A^2$: Smaller Self-Supervised ViTs Localize Better than Larger Ones","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18549","citing_title":"Advancing Vision Transformer with Enhanced Spatial Priors","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7DS5SMLSJH2YKVQN4GQRWGGH6W","json":"https://pith.science/pith/7DS5SMLSJH2YKVQN4GQRWGGH6W.json","graph_json":"https://pith.science/api/pith-number/7DS5SMLSJH2YKVQN4GQRWGGH6W/graph.json","events_json":"https://pith.science/api/pith-number/7DS5SMLSJH2YKVQN4GQRWGGH6W/events.json","paper":"https://pith.science/paper/7DS5SMLS"},"agent_actions":{"view_html":"https://pith.science/pith/7DS5SMLSJH2YKVQN4GQRWGGH6W","download_json":"https://pith.science/pith/7DS5SMLSJH2YKVQN4GQRWGGH6W.json","view_paper":"https://pith.science/paper/7DS5SMLS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.13043&json=true","fetch_graph":"https://pith.science/api/pith-number/7DS5SMLSJH2YKVQN4GQRWGGH6W/graph.json","fetch_events":"https://pith.science/api/pith-number/7DS5SMLSJH2YKVQN4GQRWGGH6W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7DS5SMLSJH2YKVQN4GQRWGGH6W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7DS5SMLSJH2YKVQN4GQRWGGH6W/action/storage_attestation","attest_author":"https://pith.science/pith/7DS5SMLSJH2YKVQN4GQRWGGH6W/action/author_attestation","sign_citation":"https://pith.science/pith/7DS5SMLSJH2YKVQN4GQRWGGH6W/action/citation_signature","submit_replication":"https://pith.science/pith/7DS5SMLSJH2YKVQN4GQRWGGH6W/action/replication_record"}},"created_at":"2026-07-05T08:45:26.459062+00:00","updated_at":"2026-07-05T08:45:26.459062+00:00"}