{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:X54ZTFUOTNMC5BY3N5P5NVKANC","short_pith_number":"pith:X54ZTFUO","schema_version":"1.0","canonical_sha256":"bf7999968e9b582e871b6f5fd6d54068b6fb438ad56e8b9c5d97232c1ed2ba24","source":{"kind":"arxiv","id":"2306.00989","version":1},"attestation_state":"computed","paper":{"title":"Hiera: A Hierarchical Vision Transformer without the Bells-and-Whistles","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Arkabandhu Chowdhury, Chaitanya Ryali, Chen Wei, Christoph Feichtenhofer, Daniel Bolya, Haoqi Fan, Jitendra Malik, Judy Hoffman, Omid Poursaeed, Po-Yao Huang, Vaibhav Aggarwal, Yanghao Li, Yuan-Ting Hu","submitted_at":"2023-06-01T17:59:58Z","abstract_excerpt":"Modern hierarchical vision transformers have added several vision-specific components in the pursuit of supervised classification performance. While these components lead to effective accuracies and attractive FLOP counts, the added complexity actually makes these transformers slower than their vanilla ViT counterparts. In this paper, we argue that this additional bulk is unnecessary. By pretraining with a strong visual pretext task (MAE), we can strip out all the bells-and-whistles from a state-of-the-art multi-stage vision transformer without losing accuracy. In the process, we create Hiera,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.00989","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-06-01T17:59:58Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"3f2c48a4bd7f767fc703f5dfb87358b83254a489f64fc1014a54e8c59604c5f0","abstract_canon_sha256":"7156f3006f7a8e9eb3f1c8b3ab1fc181cc0eb936a8c49db054dbca8eb4de6f41"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:16:34.120585Z","signature_b64":"kPup/OUz7Iwqmt5pMKmqvxvvJq9zJqi/Z0izuxm8ZgGwGShgbLL+wm10lUX67IThBMMf2KAWIeNXi5LGkiDQCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bf7999968e9b582e871b6f5fd6d54068b6fb438ad56e8b9c5d97232c1ed2ba24","last_reissued_at":"2026-07-05T06:16:34.120055Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:16:34.120055Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hiera: A Hierarchical Vision Transformer without the Bells-and-Whistles","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Arkabandhu Chowdhury, Chaitanya Ryali, Chen Wei, Christoph Feichtenhofer, Daniel Bolya, Haoqi Fan, Jitendra Malik, Judy Hoffman, Omid Poursaeed, Po-Yao Huang, Vaibhav Aggarwal, Yanghao Li, Yuan-Ting Hu","submitted_at":"2023-06-01T17:59:58Z","abstract_excerpt":"Modern hierarchical vision transformers have added several vision-specific components in the pursuit of supervised classification performance. While these components lead to effective accuracies and attractive FLOP counts, the added complexity actually makes these transformers slower than their vanilla ViT counterparts. In this paper, we argue that this additional bulk is unnecessary. By pretraining with a strong visual pretext task (MAE), we can strip out all the bells-and-whistles from a state-of-the-art multi-stage vision transformer without losing accuracy. In the process, we create Hiera,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.00989","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.00989/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.00989","created_at":"2026-07-05T06:16:34.120116+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.00989v1","created_at":"2026-07-05T06:16:34.120116+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.00989","created_at":"2026-07-05T06:16:34.120116+00:00"},{"alias_kind":"pith_short_12","alias_value":"X54ZTFUOTNMC","created_at":"2026-07-05T06:16:34.120116+00:00"},{"alias_kind":"pith_short_16","alias_value":"X54ZTFUOTNMC5BY3","created_at":"2026-07-05T06:16:34.120116+00:00"},{"alias_kind":"pith_short_8","alias_value":"X54ZTFUO","created_at":"2026-07-05T06:16:34.120116+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09162","citing_title":"Zero-Parameter Geometric Gating for Temporally Stable Low-Altitude UAV Video Semantic Segmentation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2404.08471","citing_title":"Revisiting Feature Prediction for Learning Visual Representations from Video","ref_index":270,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X54ZTFUOTNMC5BY3N5P5NVKANC","json":"https://pith.science/pith/X54ZTFUOTNMC5BY3N5P5NVKANC.json","graph_json":"https://pith.science/api/pith-number/X54ZTFUOTNMC5BY3N5P5NVKANC/graph.json","events_json":"https://pith.science/api/pith-number/X54ZTFUOTNMC5BY3N5P5NVKANC/events.json","paper":"https://pith.science/paper/X54ZTFUO"},"agent_actions":{"view_html":"https://pith.science/pith/X54ZTFUOTNMC5BY3N5P5NVKANC","download_json":"https://pith.science/pith/X54ZTFUOTNMC5BY3N5P5NVKANC.json","view_paper":"https://pith.science/paper/X54ZTFUO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.00989&json=true","fetch_graph":"https://pith.science/api/pith-number/X54ZTFUOTNMC5BY3N5P5NVKANC/graph.json","fetch_events":"https://pith.science/api/pith-number/X54ZTFUOTNMC5BY3N5P5NVKANC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X54ZTFUOTNMC5BY3N5P5NVKANC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X54ZTFUOTNMC5BY3N5P5NVKANC/action/storage_attestation","attest_author":"https://pith.science/pith/X54ZTFUOTNMC5BY3N5P5NVKANC/action/author_attestation","sign_citation":"https://pith.science/pith/X54ZTFUOTNMC5BY3N5P5NVKANC/action/citation_signature","submit_replication":"https://pith.science/pith/X54ZTFUOTNMC5BY3N5P5NVKANC/action/replication_record"}},"created_at":"2026-07-05T06:16:34.120116+00:00","updated_at":"2026-07-05T06:16:34.120116+00:00"}