{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:JFNEWJQUQDR4WMO24ARH22JTDB","short_pith_number":"pith:JFNEWJQU","schema_version":"1.0","canonical_sha256":"495a4b261480e3cb31dae0227d6933184e7cdb15145c9feabaf450504bac5ccc","source":{"kind":"arxiv","id":"2006.15055","version":2},"attestation_state":"computed","paper":{"title":"Object-Centric Learning with Slot Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexey Dosovitskiy, Aravindh Mahendran, Dirk Weissenborn, Francesco Locatello, Georg Heigold, Jakob Uszkoreit, Thomas Kipf, Thomas Unterthiner","submitted_at":"2020-06-26T15:31:57Z","abstract_excerpt":"Learning object-centric representations of complex scenes is a promising step towards enabling efficient abstract reasoning from low-level perceptual features. Yet, most deep learning approaches learn distributed representations that do not capture the compositional properties of natural scenes. In this paper, we present the Slot Attention module, an architectural component that interfaces with perceptual representations such as the output of a convolutional neural network and produces a set of task-dependent abstract representations which we call slots. These slots are exchangeable and can bi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.15055","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-26T15:31:57Z","cross_cats_sorted":["cs.CV","stat.ML"],"title_canon_sha256":"30a893ac2e2d41d8650393d9b498c9ec430b78ff143b9c54b0319c616063871a","abstract_canon_sha256":"55e43f8ef99bb06b21de01f4d52c65fda4cd0e26bfbe8a95ca94464973a6aeaf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:42:55.135611Z","signature_b64":"5G7y6qeO1T8UGDiwbWsVWyjU5JVxT+vje8BQ2hXaeQhEMxpOUTILPKXM2zPKOJgYlqVbwV76K7VUK4Q7+fTKAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"495a4b261480e3cb31dae0227d6933184e7cdb15145c9feabaf450504bac5ccc","last_reissued_at":"2026-07-05T01:42:55.135158Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:42:55.135158Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Object-Centric Learning with Slot Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexey Dosovitskiy, Aravindh Mahendran, Dirk Weissenborn, Francesco Locatello, Georg Heigold, Jakob Uszkoreit, Thomas Kipf, Thomas Unterthiner","submitted_at":"2020-06-26T15:31:57Z","abstract_excerpt":"Learning object-centric representations of complex scenes is a promising step towards enabling efficient abstract reasoning from low-level perceptual features. Yet, most deep learning approaches learn distributed representations that do not capture the compositional properties of natural scenes. In this paper, we present the Slot Attention module, an architectural component that interfaces with perceptual representations such as the output of a convolutional neural network and produces a set of task-dependent abstract representations which we call slots. These slots are exchangeable and can bi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.15055","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.15055/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.15055","created_at":"2026-07-05T01:42:55.135215+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.15055v2","created_at":"2026-07-05T01:42:55.135215+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.15055","created_at":"2026-07-05T01:42:55.135215+00:00"},{"alias_kind":"pith_short_12","alias_value":"JFNEWJQUQDR4","created_at":"2026-07-05T01:42:55.135215+00:00"},{"alias_kind":"pith_short_16","alias_value":"JFNEWJQUQDR4WMO2","created_at":"2026-07-05T01:42:55.135215+00:00"},{"alias_kind":"pith_short_8","alias_value":"JFNEWJQU","created_at":"2026-07-05T01:42:55.135215+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.06925","citing_title":"Grounding Spatial Relations in a Compact World Model: Instruction Leakage and a Goal-Free Dynamics Fix","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06014","citing_title":"Escaping the Procrustean Bed: Groupwise Orthogonal Connectors for Audio-Language Models","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2606.08032","citing_title":"Variational Proximal Policy Optimization","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05328","citing_title":"The Invisible Hand of Physics: When Video Diffusion Models Know More Than They Show","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30542","citing_title":"Physically Viable World Models: A Case for Query-Conditioned Embodied AI","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23259","citing_title":"Multi-Gate Residuals","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11654","citing_title":"Weather-Robust Cross-View Geo-Localization via Prototype-Based Semantic Part Discovery","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11654","citing_title":"Weather-Robust Cross-View Geo-Localization via Prototype-Based Semantic Part Discovery","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06481","citing_title":"OA-WAM: Object-Addressable World Action Model for Robust Robot Manipulation","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19683","citing_title":"Mask World Model: Predicting What Matters for Robust Robot Policy Learning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20925","citing_title":"Unsupervised Learning of Inter-Object Relationships via Group Homomorphism","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JFNEWJQUQDR4WMO24ARH22JTDB","json":"https://pith.science/pith/JFNEWJQUQDR4WMO24ARH22JTDB.json","graph_json":"https://pith.science/api/pith-number/JFNEWJQUQDR4WMO24ARH22JTDB/graph.json","events_json":"https://pith.science/api/pith-number/JFNEWJQUQDR4WMO24ARH22JTDB/events.json","paper":"https://pith.science/paper/JFNEWJQU"},"agent_actions":{"view_html":"https://pith.science/pith/JFNEWJQUQDR4WMO24ARH22JTDB","download_json":"https://pith.science/pith/JFNEWJQUQDR4WMO24ARH22JTDB.json","view_paper":"https://pith.science/paper/JFNEWJQU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.15055&json=true","fetch_graph":"https://pith.science/api/pith-number/JFNEWJQUQDR4WMO24ARH22JTDB/graph.json","fetch_events":"https://pith.science/api/pith-number/JFNEWJQUQDR4WMO24ARH22JTDB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JFNEWJQUQDR4WMO24ARH22JTDB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JFNEWJQUQDR4WMO24ARH22JTDB/action/storage_attestation","attest_author":"https://pith.science/pith/JFNEWJQUQDR4WMO24ARH22JTDB/action/author_attestation","sign_citation":"https://pith.science/pith/JFNEWJQUQDR4WMO24ARH22JTDB/action/citation_signature","submit_replication":"https://pith.science/pith/JFNEWJQUQDR4WMO24ARH22JTDB/action/replication_record"}},"created_at":"2026-07-05T01:42:55.135215+00:00","updated_at":"2026-07-05T01:42:55.135215+00:00"}