{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:K6TWWFUEFKQ7MVUMKQZMRDN2IW","short_pith_number":"pith:K6TWWFUE","schema_version":"1.0","canonical_sha256":"57a76b16842aa1f6568c5432c88dba45aeb21c8779393d8c57499b5dd16ddd9a","source":{"kind":"arxiv","id":"2111.12594","version":2},"attestation_state":"computed","paper":{"title":"Conditional Object-Centric Learning from Video","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CV","authors_text":"Alexey Dosovitskiy, Aravindh Mahendran, Austin Stone, Gamaleldin F. Elsayed, Georg Heigold, Klaus Greff, Rico Jonschkowski, Sara Sabour, Thomas Kipf","submitted_at":"2021-11-24T16:10:46Z","abstract_excerpt":"Object-centric representations are a promising path toward more systematic generalization by providing flexible abstractions upon which compositional world models can be built. Recent work on simple 2D and 3D datasets has shown that models with object-centric inductive biases can learn to segment and represent meaningful objects from the statistical structure of the data alone without the need for any supervision. However, such fully-unsupervised methods still fail to scale to diverse realistic data, despite the use of increasingly complex inductive biases such as priors for the size of object"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.12594","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-11-24T16:10:46Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"feb249ccc856ccfdf1c0def760bab6e1f4aebe8930135f74a710834c0353602b","abstract_canon_sha256":"9fa5037a9e5795aef3cc1cd6e82577ed7796f90018bee9c691ce7babb8ae5be8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:05:14.817929Z","signature_b64":"/VuD0xBlIXAJkzjI0gCQKzdVW/XqlytjTzMXQWD4OBww3yGMEzDfHAynKHTmoj2e9C+vlLrqshicT3DVRcXfAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57a76b16842aa1f6568c5432c88dba45aeb21c8779393d8c57499b5dd16ddd9a","last_reissued_at":"2026-07-05T04:05:14.817537Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:05:14.817537Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Conditional Object-Centric Learning from Video","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CV","authors_text":"Alexey Dosovitskiy, Aravindh Mahendran, Austin Stone, Gamaleldin F. Elsayed, Georg Heigold, Klaus Greff, Rico Jonschkowski, Sara Sabour, Thomas Kipf","submitted_at":"2021-11-24T16:10:46Z","abstract_excerpt":"Object-centric representations are a promising path toward more systematic generalization by providing flexible abstractions upon which compositional world models can be built. Recent work on simple 2D and 3D datasets has shown that models with object-centric inductive biases can learn to segment and represent meaningful objects from the statistical structure of the data alone without the need for any supervision. However, such fully-unsupervised methods still fail to scale to diverse realistic data, despite the use of increasingly complex inductive biases such as priors for the size of object"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.12594","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.12594/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.12594","created_at":"2026-07-05T04:05:14.817605+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.12594v2","created_at":"2026-07-05T04:05:14.817605+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.12594","created_at":"2026-07-05T04:05:14.817605+00:00"},{"alias_kind":"pith_short_12","alias_value":"K6TWWFUEFKQ7","created_at":"2026-07-05T04:05:14.817605+00:00"},{"alias_kind":"pith_short_16","alias_value":"K6TWWFUEFKQ7MVUM","created_at":"2026-07-05T04:05:14.817605+00:00"},{"alias_kind":"pith_short_8","alias_value":"K6TWWFUE","created_at":"2026-07-05T04:05:14.817605+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12601","citing_title":"Dual-State Slot Attention: Decoupling Appearance and Identity for Video Object-Centric Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06481","citing_title":"OA-WAM: Object-Addressable World Action Model for Robust Robot Manipulation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09045","citing_title":"Scene-Agnostic Object-Centric Representation Learning for 3D Gaussian Splatting","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07099","citing_title":"InfoGeo: Information-Theoretic Object-Centric Learning for Cross-View Generalizable UAV Geo-Localization","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17876","citing_title":"OFlow: Injecting Object-Aware Temporal Flow Matching for Robust Robotic Manipulation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20925","citing_title":"Unsupervised Learning of Inter-Object Relationships via Group Homomorphism","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K6TWWFUEFKQ7MVUMKQZMRDN2IW","json":"https://pith.science/pith/K6TWWFUEFKQ7MVUMKQZMRDN2IW.json","graph_json":"https://pith.science/api/pith-number/K6TWWFUEFKQ7MVUMKQZMRDN2IW/graph.json","events_json":"https://pith.science/api/pith-number/K6TWWFUEFKQ7MVUMKQZMRDN2IW/events.json","paper":"https://pith.science/paper/K6TWWFUE"},"agent_actions":{"view_html":"https://pith.science/pith/K6TWWFUEFKQ7MVUMKQZMRDN2IW","download_json":"https://pith.science/pith/K6TWWFUEFKQ7MVUMKQZMRDN2IW.json","view_paper":"https://pith.science/paper/K6TWWFUE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.12594&json=true","fetch_graph":"https://pith.science/api/pith-number/K6TWWFUEFKQ7MVUMKQZMRDN2IW/graph.json","fetch_events":"https://pith.science/api/pith-number/K6TWWFUEFKQ7MVUMKQZMRDN2IW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K6TWWFUEFKQ7MVUMKQZMRDN2IW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K6TWWFUEFKQ7MVUMKQZMRDN2IW/action/storage_attestation","attest_author":"https://pith.science/pith/K6TWWFUEFKQ7MVUMKQZMRDN2IW/action/author_attestation","sign_citation":"https://pith.science/pith/K6TWWFUEFKQ7MVUMKQZMRDN2IW/action/citation_signature","submit_replication":"https://pith.science/pith/K6TWWFUEFKQ7MVUMKQZMRDN2IW/action/replication_record"}},"created_at":"2026-07-05T04:05:14.817605+00:00","updated_at":"2026-07-05T04:05:14.817605+00:00"}