{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:R3UTTCBFMIBZUHHHNOECAAFREF","short_pith_number":"pith:R3UTTCBF","schema_version":"1.0","canonical_sha256":"8ee939882562039a1ce76b882000b12147e6d715593e3d4b14d2e18b7e63d915","source":{"kind":"arxiv","id":"2206.00790","version":3},"attestation_state":"computed","paper":{"title":"Efficient Self-supervised Vision Pretraining with Local Masked Reconstruction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Boyang Li, Jun Chen, Ming Hu, Mohamed Elhoseiny","submitted_at":"2022-06-01T22:46:34Z","abstract_excerpt":"Self-supervised learning for computer vision has achieved tremendous progress and improved many downstream vision tasks such as image classification, semantic segmentation, and object detection. Among these, generative self-supervised vision learning approaches such as MAE and BEiT show promising performance. However, their global masked reconstruction mechanism is computationally demanding. To address this issue, we propose local masked reconstruction (LoMaR), a simple yet effective approach that performs masked reconstruction within a small window of 7$\\times$7 patches on a simple Transforme"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.00790","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-06-01T22:46:34Z","cross_cats_sorted":[],"title_canon_sha256":"eafee4b8d929c5f8cc9252066fb810e10f669c507d31f2477c2be4ab47569acf","abstract_canon_sha256":"62148fe278af8228979d6ca4ac0cb4dd511fb6dc9dfac646f926e4ffd76ad7de"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:37:18.824469Z","signature_b64":"19plyZPd5K77fc6UHVX/GIIob5EljCeV4QjA76R3Mz4jR7jR3Tj26xVB5S/1oNe5OFtdv6YQXNviw4U+2sWUBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ee939882562039a1ce76b882000b12147e6d715593e3d4b14d2e18b7e63d915","last_reissued_at":"2026-07-05T10:37:18.823823Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:37:18.823823Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Self-supervised Vision Pretraining with Local Masked Reconstruction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Boyang Li, Jun Chen, Ming Hu, Mohamed Elhoseiny","submitted_at":"2022-06-01T22:46:34Z","abstract_excerpt":"Self-supervised learning for computer vision has achieved tremendous progress and improved many downstream vision tasks such as image classification, semantic segmentation, and object detection. Among these, generative self-supervised vision learning approaches such as MAE and BEiT show promising performance. However, their global masked reconstruction mechanism is computationally demanding. To address this issue, we propose local masked reconstruction (LoMaR), a simple yet effective approach that performs masked reconstruction within a small window of 7$\\times$7 patches on a simple Transforme"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.00790","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.00790/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.00790","created_at":"2026-07-05T10:37:18.823899+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.00790v3","created_at":"2026-07-05T10:37:18.823899+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.00790","created_at":"2026-07-05T10:37:18.823899+00:00"},{"alias_kind":"pith_short_12","alias_value":"R3UTTCBFMIBZ","created_at":"2026-07-05T10:37:18.823899+00:00"},{"alias_kind":"pith_short_16","alias_value":"R3UTTCBFMIBZUHHH","created_at":"2026-07-05T10:37:18.823899+00:00"},{"alias_kind":"pith_short_8","alias_value":"R3UTTCBF","created_at":"2026-07-05T10:37:18.823899+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.15746","citing_title":"PR-MIM: Delving Deeper into Partial Reconstruction in Masked Image Modeling","ref_index":6,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R3UTTCBFMIBZUHHHNOECAAFREF","json":"https://pith.science/pith/R3UTTCBFMIBZUHHHNOECAAFREF.json","graph_json":"https://pith.science/api/pith-number/R3UTTCBFMIBZUHHHNOECAAFREF/graph.json","events_json":"https://pith.science/api/pith-number/R3UTTCBFMIBZUHHHNOECAAFREF/events.json","paper":"https://pith.science/paper/R3UTTCBF"},"agent_actions":{"view_html":"https://pith.science/pith/R3UTTCBFMIBZUHHHNOECAAFREF","download_json":"https://pith.science/pith/R3UTTCBFMIBZUHHHNOECAAFREF.json","view_paper":"https://pith.science/paper/R3UTTCBF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.00790&json=true","fetch_graph":"https://pith.science/api/pith-number/R3UTTCBFMIBZUHHHNOECAAFREF/graph.json","fetch_events":"https://pith.science/api/pith-number/R3UTTCBFMIBZUHHHNOECAAFREF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R3UTTCBFMIBZUHHHNOECAAFREF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R3UTTCBFMIBZUHHHNOECAAFREF/action/storage_attestation","attest_author":"https://pith.science/pith/R3UTTCBFMIBZUHHHNOECAAFREF/action/author_attestation","sign_citation":"https://pith.science/pith/R3UTTCBFMIBZUHHHNOECAAFREF/action/citation_signature","submit_replication":"https://pith.science/pith/R3UTTCBFMIBZUHHHNOECAAFREF/action/replication_record"}},"created_at":"2026-07-05T10:37:18.823899+00:00","updated_at":"2026-07-05T10:37:18.823899+00:00"}