{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:Y4QHYMHTO5L3MZ54NSQRVFOHXQ","short_pith_number":"pith:Y4QHYMHT","schema_version":"1.0","canonical_sha256":"c7207c30f37757b667bc6ca11a95c7bc0c906b4ab1b2cbe76e907c8d536d49fd","source":{"kind":"arxiv","id":"2202.03382","version":2},"attestation_state":"computed","paper":{"title":"Corrupted Image Modeling for Self-Supervised Visual Pre-Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Furu Wei, Hangbo Bao, Li Dong, Xinggang Wang, Yuxin Fang","submitted_at":"2022-02-07T17:59:04Z","abstract_excerpt":"We introduce Corrupted Image Modeling (CIM) for self-supervised visual pre-training. CIM uses an auxiliary generator with a small trainable BEiT to corrupt the input image instead of using artificial [MASK] tokens, where some patches are randomly selected and replaced with plausible alternatives sampled from the BEiT output distribution. Given this corrupted image, an enhancer network learns to either recover all the original image pixels, or predict whether each visual token is replaced by a generator sample or not. The generator and the enhancer are simultaneously trained and synergistically"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.03382","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-02-07T17:59:04Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"03ddde802a9dcb1247d02f039cd187b8ffe0b90da0d63fa0bc022030a046267c","abstract_canon_sha256":"f731ddd044b96221b3371a36e6aaf3883e5dd5600814892101636494553828b5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:51:39.401424Z","signature_b64":"tx4zB1xM9HVj0KYEyhlf6b6lgAxNMJquXoql1/SEM5HT16ifUm/0W1hcF5GEdL4YpGXvdjXgG/UC8hppXDWrCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c7207c30f37757b667bc6ca11a95c7bc0c906b4ab1b2cbe76e907c8d536d49fd","last_reissued_at":"2026-07-05T05:51:39.400986Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:51:39.400986Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Corrupted Image Modeling for Self-Supervised Visual Pre-Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Furu Wei, Hangbo Bao, Li Dong, Xinggang Wang, Yuxin Fang","submitted_at":"2022-02-07T17:59:04Z","abstract_excerpt":"We introduce Corrupted Image Modeling (CIM) for self-supervised visual pre-training. CIM uses an auxiliary generator with a small trainable BEiT to corrupt the input image instead of using artificial [MASK] tokens, where some patches are randomly selected and replaced with plausible alternatives sampled from the BEiT output distribution. Given this corrupted image, an enhancer network learns to either recover all the original image pixels, or predict whether each visual token is replaced by a generator sample or not. The generator and the enhancer are simultaneously trained and synergistically"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.03382","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.03382/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.03382","created_at":"2026-07-05T05:51:39.401044+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.03382v2","created_at":"2026-07-05T05:51:39.401044+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.03382","created_at":"2026-07-05T05:51:39.401044+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y4QHYMHTO5L3","created_at":"2026-07-05T05:51:39.401044+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y4QHYMHTO5L3MZ54","created_at":"2026-07-05T05:51:39.401044+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y4QHYMHT","created_at":"2026-07-05T05:51:39.401044+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.09072","citing_title":"Cross-View Completion Models are Zero-shot Correspondence Estimators","ref_index":23,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y4QHYMHTO5L3MZ54NSQRVFOHXQ","json":"https://pith.science/pith/Y4QHYMHTO5L3MZ54NSQRVFOHXQ.json","graph_json":"https://pith.science/api/pith-number/Y4QHYMHTO5L3MZ54NSQRVFOHXQ/graph.json","events_json":"https://pith.science/api/pith-number/Y4QHYMHTO5L3MZ54NSQRVFOHXQ/events.json","paper":"https://pith.science/paper/Y4QHYMHT"},"agent_actions":{"view_html":"https://pith.science/pith/Y4QHYMHTO5L3MZ54NSQRVFOHXQ","download_json":"https://pith.science/pith/Y4QHYMHTO5L3MZ54NSQRVFOHXQ.json","view_paper":"https://pith.science/paper/Y4QHYMHT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.03382&json=true","fetch_graph":"https://pith.science/api/pith-number/Y4QHYMHTO5L3MZ54NSQRVFOHXQ/graph.json","fetch_events":"https://pith.science/api/pith-number/Y4QHYMHTO5L3MZ54NSQRVFOHXQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y4QHYMHTO5L3MZ54NSQRVFOHXQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y4QHYMHTO5L3MZ54NSQRVFOHXQ/action/storage_attestation","attest_author":"https://pith.science/pith/Y4QHYMHTO5L3MZ54NSQRVFOHXQ/action/author_attestation","sign_citation":"https://pith.science/pith/Y4QHYMHTO5L3MZ54NSQRVFOHXQ/action/citation_signature","submit_replication":"https://pith.science/pith/Y4QHYMHTO5L3MZ54NSQRVFOHXQ/action/replication_record"}},"created_at":"2026-07-05T05:51:39.401044+00:00","updated_at":"2026-07-05T05:51:39.401044+00:00"}