{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ULN3ZENF3GHUZXT54WQWTEI2GG","short_pith_number":"pith:ULN3ZENF","schema_version":"1.0","canonical_sha256":"a2dbbc91a5d98f4cde7de5a169911a318be6b49cfcc4d9ca02f07b406f588c6d","source":{"kind":"arxiv","id":"2307.12698","version":1},"attestation_state":"computed","paper":{"title":"MC-JEPA: A Joint-Embedding Predictive Architecture for Self-Supervised Learning of Motion and Content Features","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Adrien Bardes, Jean Ponce, Yann LeCun","submitted_at":"2023-07-24T11:27:14Z","abstract_excerpt":"Self-supervised learning of visual representations has been focusing on learning content features, which do not capture object motion or location, and focus on identifying and differentiating objects in images and videos. On the other hand, optical flow estimation is a task that does not involve understanding the content of the images on which it is estimated. We unify the two approaches and introduce MC-JEPA, a joint-embedding predictive architecture and self-supervised learning approach to jointly learn optical flow and content features within a shared encoder, demonstrating that the two ass"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.12698","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-07-24T11:27:14Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"016fc7faf6fca58032535fc6803a74c5696c6ecdade8e877274e0ee0e9a83a94","abstract_canon_sha256":"30e004cd34ba91159d3fff687f95fb3b0b19e6e7b1ce08316c1d3031b5711e27"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:33:57.501798Z","signature_b64":"hoLYSU2a2CplX3GCXONA0zKKX5DBIaON+0qDvRZ+914KZT05sxlyAG13ThAjXiAX0RgQ5gNwjZXdAdQCEsZtAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a2dbbc91a5d98f4cde7de5a169911a318be6b49cfcc4d9ca02f07b406f588c6d","last_reissued_at":"2026-07-05T06:33:57.501323Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:33:57.501323Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MC-JEPA: A Joint-Embedding Predictive Architecture for Self-Supervised Learning of Motion and Content Features","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Adrien Bardes, Jean Ponce, Yann LeCun","submitted_at":"2023-07-24T11:27:14Z","abstract_excerpt":"Self-supervised learning of visual representations has been focusing on learning content features, which do not capture object motion or location, and focus on identifying and differentiating objects in images and videos. On the other hand, optical flow estimation is a task that does not involve understanding the content of the images on which it is estimated. We unify the two approaches and introduce MC-JEPA, a joint-embedding predictive architecture and self-supervised learning approach to jointly learn optical flow and content features within a shared encoder, demonstrating that the two ass"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.12698","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.12698/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.12698","created_at":"2026-07-05T06:33:57.501398+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.12698v1","created_at":"2026-07-05T06:33:57.501398+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.12698","created_at":"2026-07-05T06:33:57.501398+00:00"},{"alias_kind":"pith_short_12","alias_value":"ULN3ZENF3GHU","created_at":"2026-07-05T06:33:57.501398+00:00"},{"alias_kind":"pith_short_16","alias_value":"ULN3ZENF3GHUZXT5","created_at":"2026-07-05T06:33:57.501398+00:00"},{"alias_kind":"pith_short_8","alias_value":"ULN3ZENF","created_at":"2026-07-05T06:33:57.501398+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23444","citing_title":"SkyJEPA: Learning Long-Horizon World Models for Zero-Shot Sim-to-Real Control of Quadrotors","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20781","citing_title":"World Action Models: A Survey","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01443","citing_title":"UR-JEPA: Uniform Rectifiability as a Regularizer for Joint-Embedding Predictive Architectures","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13684","citing_title":"Recurrent Video Masked Autoencoders","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2404.08471","citing_title":"Revisiting Feature Prediction for Learning Visual Representations from Video","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ULN3ZENF3GHUZXT54WQWTEI2GG","json":"https://pith.science/pith/ULN3ZENF3GHUZXT54WQWTEI2GG.json","graph_json":"https://pith.science/api/pith-number/ULN3ZENF3GHUZXT54WQWTEI2GG/graph.json","events_json":"https://pith.science/api/pith-number/ULN3ZENF3GHUZXT54WQWTEI2GG/events.json","paper":"https://pith.science/paper/ULN3ZENF"},"agent_actions":{"view_html":"https://pith.science/pith/ULN3ZENF3GHUZXT54WQWTEI2GG","download_json":"https://pith.science/pith/ULN3ZENF3GHUZXT54WQWTEI2GG.json","view_paper":"https://pith.science/paper/ULN3ZENF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.12698&json=true","fetch_graph":"https://pith.science/api/pith-number/ULN3ZENF3GHUZXT54WQWTEI2GG/graph.json","fetch_events":"https://pith.science/api/pith-number/ULN3ZENF3GHUZXT54WQWTEI2GG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ULN3ZENF3GHUZXT54WQWTEI2GG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ULN3ZENF3GHUZXT54WQWTEI2GG/action/storage_attestation","attest_author":"https://pith.science/pith/ULN3ZENF3GHUZXT54WQWTEI2GG/action/author_attestation","sign_citation":"https://pith.science/pith/ULN3ZENF3GHUZXT54WQWTEI2GG/action/citation_signature","submit_replication":"https://pith.science/pith/ULN3ZENF3GHUZXT54WQWTEI2GG/action/replication_record"}},"created_at":"2026-07-05T06:33:57.501398+00:00","updated_at":"2026-07-05T06:33:57.501398+00:00"}