{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BRWZY25YUOLFD5B73KTJLTDJY5","short_pith_number":"pith:BRWZY25Y","schema_version":"1.0","canonical_sha256":"0c6d9c6bb8a39651f43fdaa695cc69c76155ab3a4a41a2f863a2bee0a7268d89","source":{"kind":"arxiv","id":"2301.03580","version":2},"attestation_state":"computed","paper":{"title":"Designing BERT for Convolutional Networks: Sparse and Hierarchical Masked Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chen Lin, Keyu Tian, Liwei Wang, Qishuai Diao, Yi Jiang, Zehuan Yuan","submitted_at":"2023-01-09T18:59:50Z","abstract_excerpt":"We identify and overcome two key obstacles in extending the success of BERT-style pre-training, or the masked image modeling, to convolutional networks (convnets): (i) convolution operation cannot handle irregular, random-masked input images; (ii) the single-scale nature of BERT pre-training is inconsistent with convnet's hierarchical structure. For (i), we treat unmasked pixels as sparse voxels of 3D point clouds and use sparse convolution to encode. This is the first use of sparse convolution for 2D masked modeling. For (ii), we develop a hierarchical decoder to reconstruct images from multi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.03580","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-01-09T18:59:50Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"0d800c115711756b9f8960e8b16d578f7e530bf59d1d299358844f912e8fa55f","abstract_canon_sha256":"52ccb4f2fecec00b3af7ceeb4576599b6ea591b6f9e8334033f2ebb9a9b15f2e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:32:03.508828Z","signature_b64":"ON/e1+HZwxdXgRpjoxequNa0SjzjHfAdZ2oewiJovE43po2lWkb0AvpQ4uNysb4UMQq/Wf+WWjQRh4cXEThpBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c6d9c6bb8a39651f43fdaa695cc69c76155ab3a4a41a2f863a2bee0a7268d89","last_reissued_at":"2026-07-05T05:32:03.508267Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:32:03.508267Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Designing BERT for Convolutional Networks: Sparse and Hierarchical Masked Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chen Lin, Keyu Tian, Liwei Wang, Qishuai Diao, Yi Jiang, Zehuan Yuan","submitted_at":"2023-01-09T18:59:50Z","abstract_excerpt":"We identify and overcome two key obstacles in extending the success of BERT-style pre-training, or the masked image modeling, to convolutional networks (convnets): (i) convolution operation cannot handle irregular, random-masked input images; (ii) the single-scale nature of BERT pre-training is inconsistent with convnet's hierarchical structure. For (i), we treat unmasked pixels as sparse voxels of 3D point clouds and use sparse convolution to encode. This is the first use of sparse convolution for 2D masked modeling. For (ii), we develop a hierarchical decoder to reconstruct images from multi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.03580","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.03580/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.03580","created_at":"2026-07-05T05:32:03.508342+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.03580v2","created_at":"2026-07-05T05:32:03.508342+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.03580","created_at":"2026-07-05T05:32:03.508342+00:00"},{"alias_kind":"pith_short_12","alias_value":"BRWZY25YUOLF","created_at":"2026-07-05T05:32:03.508342+00:00"},{"alias_kind":"pith_short_16","alias_value":"BRWZY25YUOLFD5B7","created_at":"2026-07-05T05:32:03.508342+00:00"},{"alias_kind":"pith_short_8","alias_value":"BRWZY25Y","created_at":"2026-07-05T05:32:03.508342+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03508","citing_title":"Structure-Guided Mixed Masked Pretraining and Spatial Continuity Regularization for Printed Circuit Board Defect Detection","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08858","citing_title":"BIAS: A Biologically Inspired Algorithm for Video Saliency Detection","ref_index":89,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BRWZY25YUOLFD5B73KTJLTDJY5","json":"https://pith.science/pith/BRWZY25YUOLFD5B73KTJLTDJY5.json","graph_json":"https://pith.science/api/pith-number/BRWZY25YUOLFD5B73KTJLTDJY5/graph.json","events_json":"https://pith.science/api/pith-number/BRWZY25YUOLFD5B73KTJLTDJY5/events.json","paper":"https://pith.science/paper/BRWZY25Y"},"agent_actions":{"view_html":"https://pith.science/pith/BRWZY25YUOLFD5B73KTJLTDJY5","download_json":"https://pith.science/pith/BRWZY25YUOLFD5B73KTJLTDJY5.json","view_paper":"https://pith.science/paper/BRWZY25Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.03580&json=true","fetch_graph":"https://pith.science/api/pith-number/BRWZY25YUOLFD5B73KTJLTDJY5/graph.json","fetch_events":"https://pith.science/api/pith-number/BRWZY25YUOLFD5B73KTJLTDJY5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BRWZY25YUOLFD5B73KTJLTDJY5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BRWZY25YUOLFD5B73KTJLTDJY5/action/storage_attestation","attest_author":"https://pith.science/pith/BRWZY25YUOLFD5B73KTJLTDJY5/action/author_attestation","sign_citation":"https://pith.science/pith/BRWZY25YUOLFD5B73KTJLTDJY5/action/citation_signature","submit_replication":"https://pith.science/pith/BRWZY25YUOLFD5B73KTJLTDJY5/action/replication_record"}},"created_at":"2026-07-05T05:32:03.508342+00:00","updated_at":"2026-07-05T05:32:03.508342+00:00"}