{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TUFVU2Q3YLMPVOLX754URCWG3E","short_pith_number":"pith:TUFVU2Q3","schema_version":"1.0","canonical_sha256":"9d0b5a6a1bc2d8fab977ff79488ac6d91f1dbe700b67df24a961f14eaf817a65","source":{"kind":"arxiv","id":"2407.09087","version":1},"attestation_state":"computed","paper":{"title":"On the Role of Discrete Tokenization in Visual Representation Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Tianqi Du, Yifei Wang, Yisen Wang","submitted_at":"2024-07-12T08:25:31Z","abstract_excerpt":"In the realm of self-supervised learning (SSL), masked image modeling (MIM) has gained popularity alongside contrastive learning methods. MIM involves reconstructing masked regions of input images using their unmasked portions. A notable subset of MIM methodologies employs discrete tokens as the reconstruction target, but the theoretical underpinnings of this choice remain underexplored. In this paper, we explore the role of these discrete tokens, aiming to unravel their benefits and limitations. Building upon the connection between MIM and contrastive learning, we provide a comprehensive theo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.09087","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-12T08:25:31Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"9406830522ec89c477269f48ccbdd37a2de9851e902fcd635a70eedc751f7536","abstract_canon_sha256":"223928405f95db3a11e6a5d5225aaad5a58ed04024e106a4ffacc4d930316f71"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:14.436756Z","signature_b64":"ZQ3mvMKkEvwYjWaOxBCM+dhbrzw6fCyQkgMd/TizubjOmNtWYNyIVv0Ud3qs4mKs57rnU/JylwndoCYgkE9SAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d0b5a6a1bc2d8fab977ff79488ac6d91f1dbe700b67df24a961f14eaf817a65","last_reissued_at":"2026-07-05T08:43:14.436293Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:14.436293Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Role of Discrete Tokenization in Visual Representation Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Tianqi Du, Yifei Wang, Yisen Wang","submitted_at":"2024-07-12T08:25:31Z","abstract_excerpt":"In the realm of self-supervised learning (SSL), masked image modeling (MIM) has gained popularity alongside contrastive learning methods. MIM involves reconstructing masked regions of input images using their unmasked portions. A notable subset of MIM methodologies employs discrete tokens as the reconstruction target, but the theoretical underpinnings of this choice remain underexplored. In this paper, we explore the role of these discrete tokens, aiming to unravel their benefits and limitations. Building upon the connection between MIM and contrastive learning, we provide a comprehensive theo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.09087","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.09087/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.09087","created_at":"2026-07-05T08:43:14.436358+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.09087v1","created_at":"2026-07-05T08:43:14.436358+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.09087","created_at":"2026-07-05T08:43:14.436358+00:00"},{"alias_kind":"pith_short_12","alias_value":"TUFVU2Q3YLMP","created_at":"2026-07-05T08:43:14.436358+00:00"},{"alias_kind":"pith_short_16","alias_value":"TUFVU2Q3YLMPVOLX","created_at":"2026-07-05T08:43:14.436358+00:00"},{"alias_kind":"pith_short_8","alias_value":"TUFVU2Q3","created_at":"2026-07-05T08:43:14.436358+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.05032","citing_title":"Identifying and Understanding Cross-Class Features in Adversarial Training","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TUFVU2Q3YLMPVOLX754URCWG3E","json":"https://pith.science/pith/TUFVU2Q3YLMPVOLX754URCWG3E.json","graph_json":"https://pith.science/api/pith-number/TUFVU2Q3YLMPVOLX754URCWG3E/graph.json","events_json":"https://pith.science/api/pith-number/TUFVU2Q3YLMPVOLX754URCWG3E/events.json","paper":"https://pith.science/paper/TUFVU2Q3"},"agent_actions":{"view_html":"https://pith.science/pith/TUFVU2Q3YLMPVOLX754URCWG3E","download_json":"https://pith.science/pith/TUFVU2Q3YLMPVOLX754URCWG3E.json","view_paper":"https://pith.science/paper/TUFVU2Q3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.09087&json=true","fetch_graph":"https://pith.science/api/pith-number/TUFVU2Q3YLMPVOLX754URCWG3E/graph.json","fetch_events":"https://pith.science/api/pith-number/TUFVU2Q3YLMPVOLX754URCWG3E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TUFVU2Q3YLMPVOLX754URCWG3E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TUFVU2Q3YLMPVOLX754URCWG3E/action/storage_attestation","attest_author":"https://pith.science/pith/TUFVU2Q3YLMPVOLX754URCWG3E/action/author_attestation","sign_citation":"https://pith.science/pith/TUFVU2Q3YLMPVOLX754URCWG3E/action/citation_signature","submit_replication":"https://pith.science/pith/TUFVU2Q3YLMPVOLX754URCWG3E/action/replication_record"}},"created_at":"2026-07-05T08:43:14.436358+00:00","updated_at":"2026-07-05T08:43:14.436358+00:00"}