{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GTGNW2RUXQ2VSZ3RYSDSQL5XHJ","short_pith_number":"pith:GTGNW2RU","schema_version":"1.0","canonical_sha256":"34ccdb6a34bc35596771c487282fb73a63d1bb22a545f4652fbfc93fc5957ae9","source":{"kind":"arxiv","id":"2407.19985","version":2},"attestation_state":"computed","paper":{"title":"Mixture of Nested Experts: Adaptive Processing of Visual Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aditya Kusupati, Anurag Arnab, Arsha Nagrani, Gagan Jain, Nidhi Hegde, Prateek Jain, Shyamal Buch, Sujoy Paul","submitted_at":"2024-07-29T13:19:31Z","abstract_excerpt":"The visual medium (images and videos) naturally contains a large amount of information redundancy, thereby providing a great opportunity for leveraging efficiency in processing. While Vision Transformer (ViT) based models scale effectively to large data regimes, they fail to capitalize on this inherent redundancy, leading to higher computational costs. Mixture of Experts (MoE) networks demonstrate scalability while maintaining same inference-time costs, but they come with a larger parameter footprint. We present Mixture of Nested Experts (MoNE), which utilizes a nested structure for experts, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.19985","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-29T13:19:31Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"d103c3925a244b9b24965d620b91394e509c15ecd94a51135ce8ca13a950b064","abstract_canon_sha256":"982f9c28bc55dac8a88e16cfa377c3be92af0169fb10fb3a109d66b6daa06296"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:50:14.906477Z","signature_b64":"RvpUknyDb+t8qU2KYBRDFMM3XflANIyu3vJA2ND7pT4LprkC02hlinDP4nz/QKEf7f7XKCz7/r+XJU2TSgAjAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"34ccdb6a34bc35596771c487282fb73a63d1bb22a545f4652fbfc93fc5957ae9","last_reissued_at":"2026-07-05T08:50:14.906000Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:50:14.906000Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mixture of Nested Experts: Adaptive Processing of Visual Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aditya Kusupati, Anurag Arnab, Arsha Nagrani, Gagan Jain, Nidhi Hegde, Prateek Jain, Shyamal Buch, Sujoy Paul","submitted_at":"2024-07-29T13:19:31Z","abstract_excerpt":"The visual medium (images and videos) naturally contains a large amount of information redundancy, thereby providing a great opportunity for leveraging efficiency in processing. While Vision Transformer (ViT) based models scale effectively to large data regimes, they fail to capitalize on this inherent redundancy, leading to higher computational costs. Mixture of Experts (MoE) networks demonstrate scalability while maintaining same inference-time costs, but they come with a larger parameter footprint. We present Mixture of Nested Experts (MoNE), which utilizes a nested structure for experts, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.19985","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.19985/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.19985","created_at":"2026-07-05T08:50:14.906059+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.19985v2","created_at":"2026-07-05T08:50:14.906059+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.19985","created_at":"2026-07-05T08:50:14.906059+00:00"},{"alias_kind":"pith_short_12","alias_value":"GTGNW2RUXQ2V","created_at":"2026-07-05T08:50:14.906059+00:00"},{"alias_kind":"pith_short_16","alias_value":"GTGNW2RUXQ2VSZ3R","created_at":"2026-07-05T08:50:14.906059+00:00"},{"alias_kind":"pith_short_8","alias_value":"GTGNW2RU","created_at":"2026-07-05T08:50:14.906059+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.06313","citing_title":"Surrogate-Enhanced Modeling and Adaptive Modular Control of All-Electric Heavy-Duty Robotic Manipulators","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ","json":"https://pith.science/pith/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ.json","graph_json":"https://pith.science/api/pith-number/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ/graph.json","events_json":"https://pith.science/api/pith-number/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ/events.json","paper":"https://pith.science/paper/GTGNW2RU"},"agent_actions":{"view_html":"https://pith.science/pith/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ","download_json":"https://pith.science/pith/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ.json","view_paper":"https://pith.science/paper/GTGNW2RU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.19985&json=true","fetch_graph":"https://pith.science/api/pith-number/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ/graph.json","fetch_events":"https://pith.science/api/pith-number/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ/action/storage_attestation","attest_author":"https://pith.science/pith/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ/action/author_attestation","sign_citation":"https://pith.science/pith/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ/action/citation_signature","submit_replication":"https://pith.science/pith/GTGNW2RUXQ2VSZ3RYSDSQL5XHJ/action/replication_record"}},"created_at":"2026-07-05T08:50:14.906059+00:00","updated_at":"2026-07-05T08:50:14.906059+00:00"}