{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LDLQHR4BYMMK5MPELK2XBUZURK","short_pith_number":"pith:LDLQHR4B","schema_version":"1.0","canonical_sha256":"58d703c781c318aeb1e45ab570d3348aa9faa622c5c9d381111af534eb97e204","source":{"kind":"arxiv","id":"2403.00504","version":1},"attestation_state":"computed","paper":{"title":"Learning and Leveraging World Models in Visual Representation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Adrien Bardes, Laurent Najman, Mahmoud Assran, Nicolas Ballas, Quentin Garrido, Yann LeCun","submitted_at":"2024-03-01T13:05:38Z","abstract_excerpt":"Joint-Embedding Predictive Architecture (JEPA) has emerged as a promising self-supervised approach that learns by leveraging a world model. While previously limited to predicting missing parts of an input, we explore how to generalize the JEPA prediction task to a broader set of corruptions. We introduce Image World Models, an approach that goes beyond masked image modeling and learns to predict the effect of global photometric transformations in latent space. We study the recipe of learning performant IWMs and show that it relies on three key aspects: conditioning, prediction difficulty, and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.00504","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-01T13:05:38Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"13e8f5a72ffeb4bf0673f4540cc5f7e3a5d890f485b1c3f6a7720639c8d02600","abstract_canon_sha256":"69d0f72513d4ff5c9964b61706b6b58f0a1d1489400611b1e9141275dd9c34dc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:51:05.100637Z","signature_b64":"lZOaYnnoHPOcvFrMSakCsb2K+ZJFs5ui4Khbl7ggwkbCBm1juE/2zW9Y+OxL3d2DbrUe16rSkmorDkB2xtHPBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"58d703c781c318aeb1e45ab570d3348aa9faa622c5c9d381111af534eb97e204","last_reissued_at":"2026-07-05T07:51:05.100140Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:51:05.100140Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning and Leveraging World Models in Visual Representation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Adrien Bardes, Laurent Najman, Mahmoud Assran, Nicolas Ballas, Quentin Garrido, Yann LeCun","submitted_at":"2024-03-01T13:05:38Z","abstract_excerpt":"Joint-Embedding Predictive Architecture (JEPA) has emerged as a promising self-supervised approach that learns by leveraging a world model. While previously limited to predicting missing parts of an input, we explore how to generalize the JEPA prediction task to a broader set of corruptions. We introduce Image World Models, an approach that goes beyond masked image modeling and learns to predict the effect of global photometric transformations in latent space. We study the recipe of learning performant IWMs and show that it relies on three key aspects: conditioning, prediction difficulty, and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.00504","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.00504/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.00504","created_at":"2026-07-05T07:51:05.100191+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.00504v1","created_at":"2026-07-05T07:51:05.100191+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.00504","created_at":"2026-07-05T07:51:05.100191+00:00"},{"alias_kind":"pith_short_12","alias_value":"LDLQHR4BYMMK","created_at":"2026-07-05T07:51:05.100191+00:00"},{"alias_kind":"pith_short_16","alias_value":"LDLQHR4BYMMK5MPE","created_at":"2026-07-05T07:51:05.100191+00:00"},{"alias_kind":"pith_short_8","alias_value":"LDLQHR4B","created_at":"2026-07-05T07:51:05.100191+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26282","citing_title":"Scaling World-Model Reinforcement Learning Through Diffusion Policy Optimization","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26478","citing_title":"Efficient On-policy Visual-RL via Stochastic Decoupled Policy Gradient","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21800","citing_title":"stable-worldmodel: A Platform for Reproducible World Modeling Research and Evaluation","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2602.23058","citing_title":"GeoWorld: Geometric World Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2506.10137","citing_title":"Self-Predictive Representations for Combinatorial Generalization in Behavioral Cloning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2603.07083","citing_title":"Dreamer-CDP: Improving Reconstruction-free World Models Via Continuous Deterministic Representation Prediction","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03245","citing_title":"Text-Conditional JEPA for Learning Semantically Rich Visual Representations","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05586","citing_title":"AeroJEPA: Learning Semantic Latent Representations for Scalable 3D Aerodynamic Field Modeling","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07199","citing_title":"Three-in-One World Model: Energy-Based Consistency, Prediction, and Counterfactual Inference for Marketing Intervention","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06155","citing_title":"Toward Consistent World Models with Multi-Token Prediction and Latent Semantic Enhancement","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LDLQHR4BYMMK5MPELK2XBUZURK","json":"https://pith.science/pith/LDLQHR4BYMMK5MPELK2XBUZURK.json","graph_json":"https://pith.science/api/pith-number/LDLQHR4BYMMK5MPELK2XBUZURK/graph.json","events_json":"https://pith.science/api/pith-number/LDLQHR4BYMMK5MPELK2XBUZURK/events.json","paper":"https://pith.science/paper/LDLQHR4B"},"agent_actions":{"view_html":"https://pith.science/pith/LDLQHR4BYMMK5MPELK2XBUZURK","download_json":"https://pith.science/pith/LDLQHR4BYMMK5MPELK2XBUZURK.json","view_paper":"https://pith.science/paper/LDLQHR4B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.00504&json=true","fetch_graph":"https://pith.science/api/pith-number/LDLQHR4BYMMK5MPELK2XBUZURK/graph.json","fetch_events":"https://pith.science/api/pith-number/LDLQHR4BYMMK5MPELK2XBUZURK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LDLQHR4BYMMK5MPELK2XBUZURK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LDLQHR4BYMMK5MPELK2XBUZURK/action/storage_attestation","attest_author":"https://pith.science/pith/LDLQHR4BYMMK5MPELK2XBUZURK/action/author_attestation","sign_citation":"https://pith.science/pith/LDLQHR4BYMMK5MPELK2XBUZURK/action/citation_signature","submit_replication":"https://pith.science/pith/LDLQHR4BYMMK5MPELK2XBUZURK/action/replication_record"}},"created_at":"2026-07-05T07:51:05.100191+00:00","updated_at":"2026-07-05T07:51:05.100191+00:00"}