{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OFN77I5ZAZEKUNPSSSBEJR45K7","short_pith_number":"pith:OFN77I5Z","schema_version":"1.0","canonical_sha256":"715bffa3b90648aa35f2948244c79d57e38d93c9a147d6e491489ee411b68bd7","source":{"kind":"arxiv","id":"2503.21770","version":1},"attestation_state":"computed","paper":{"title":"Visual Jenga: Discovering Object Dependencies via Counterfactual Inpainting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alexei A. Efros, Anand Bhattad, Konpat Preechakul","submitted_at":"2025-03-27T17:59:33Z","abstract_excerpt":"This paper proposes a novel scene understanding task called Visual Jenga. Drawing inspiration from the game Jenga, the proposed task involves progressively removing objects from a single image until only the background remains. Just as Jenga players must understand structural dependencies to maintain tower stability, our task reveals the intrinsic relationships between scene elements by systematically exploring which objects can be removed while preserving scene coherence in both physical and geometric sense. As a starting point for tackling the Visual Jenga task, we propose a simple, data-dri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.21770","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-27T17:59:33Z","cross_cats_sorted":[],"title_canon_sha256":"556cc1b1820b81fc1cac5186a7f6ffc5563ed7415b97d0bbae638ee89694b5a6","abstract_canon_sha256":"1b2ffb2943249aae5c51ad41126c9ff0ee8fdfdf6b6b681305d45659104f772b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:30.423638Z","signature_b64":"cA2BcWMHQDxi7B++kKrwkmxY1MS6QqrXoE9EGl2jDAiGjsfqNhCie2fLQwrqg67QEQo7wYRcBc27RBckqsAiBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"715bffa3b90648aa35f2948244c79d57e38d93c9a147d6e491489ee411b68bd7","last_reissued_at":"2026-07-05T10:40:30.423166Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:30.423166Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Jenga: Discovering Object Dependencies via Counterfactual Inpainting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alexei A. Efros, Anand Bhattad, Konpat Preechakul","submitted_at":"2025-03-27T17:59:33Z","abstract_excerpt":"This paper proposes a novel scene understanding task called Visual Jenga. Drawing inspiration from the game Jenga, the proposed task involves progressively removing objects from a single image until only the background remains. Just as Jenga players must understand structural dependencies to maintain tower stability, our task reveals the intrinsic relationships between scene elements by systematically exploring which objects can be removed while preserving scene coherence in both physical and geometric sense. As a starting point for tackling the Visual Jenga task, we propose a simple, data-dri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.21770","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.21770/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.21770","created_at":"2026-07-05T10:40:30.423227+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.21770v1","created_at":"2026-07-05T10:40:30.423227+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.21770","created_at":"2026-07-05T10:40:30.423227+00:00"},{"alias_kind":"pith_short_12","alias_value":"OFN77I5ZAZEK","created_at":"2026-07-05T10:40:30.423227+00:00"},{"alias_kind":"pith_short_16","alias_value":"OFN77I5ZAZEKUNPS","created_at":"2026-07-05T10:40:30.423227+00:00"},{"alias_kind":"pith_short_8","alias_value":"OFN77I5Z","created_at":"2026-07-05T10:40:30.423227+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00439","citing_title":"Physical Object Understanding with a Physically Controllable World Model","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2506.21546","citing_title":"Counterfactual Segmentation Reasoning: Diagnosing and Mitigating Pixel-Grounding Hallucination","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13609","citing_title":"Do-Undo Bench: Reversibility for Action Understanding in Image Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20328","citing_title":"Video models are zero-shot learners and reasoners","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00799","citing_title":"Multimodal Language Models Cannot Spot Spatial Inconsistencies","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OFN77I5ZAZEKUNPSSSBEJR45K7","json":"https://pith.science/pith/OFN77I5ZAZEKUNPSSSBEJR45K7.json","graph_json":"https://pith.science/api/pith-number/OFN77I5ZAZEKUNPSSSBEJR45K7/graph.json","events_json":"https://pith.science/api/pith-number/OFN77I5ZAZEKUNPSSSBEJR45K7/events.json","paper":"https://pith.science/paper/OFN77I5Z"},"agent_actions":{"view_html":"https://pith.science/pith/OFN77I5ZAZEKUNPSSSBEJR45K7","download_json":"https://pith.science/pith/OFN77I5ZAZEKUNPSSSBEJR45K7.json","view_paper":"https://pith.science/paper/OFN77I5Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.21770&json=true","fetch_graph":"https://pith.science/api/pith-number/OFN77I5ZAZEKUNPSSSBEJR45K7/graph.json","fetch_events":"https://pith.science/api/pith-number/OFN77I5ZAZEKUNPSSSBEJR45K7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OFN77I5ZAZEKUNPSSSBEJR45K7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OFN77I5ZAZEKUNPSSSBEJR45K7/action/storage_attestation","attest_author":"https://pith.science/pith/OFN77I5ZAZEKUNPSSSBEJR45K7/action/author_attestation","sign_citation":"https://pith.science/pith/OFN77I5ZAZEKUNPSSSBEJR45K7/action/citation_signature","submit_replication":"https://pith.science/pith/OFN77I5ZAZEKUNPSSSBEJR45K7/action/replication_record"}},"created_at":"2026-07-05T10:40:30.423227+00:00","updated_at":"2026-07-05T10:40:30.423227+00:00"}