{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:U6JXW7YEJDLQX4M3AFONYF62YZ","short_pith_number":"pith:U6JXW7YE","schema_version":"1.0","canonical_sha256":"a7937b7f0448d70bf19b015cdc17dac64d589ad3a7d8e89bb0d57dcc1d8302d8","source":{"kind":"arxiv","id":"2505.02836","version":1},"attestation_state":"computed","paper":{"title":"Scenethesis: A Language and Vision Agentic Framework for 3D Scene Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aniket Bera, Chen-Hsuan Lin, Lu Ling, Ming-Yu Liu, Tsung-Yi Lin, Yichen Sheng, Yifan Ding, Yunhao Ge, Yu Zeng, Zhaoshuo Li","submitted_at":"2025-05-05T17:59:58Z","abstract_excerpt":"Synthesizing interactive 3D scenes from text is essential for gaming, virtual reality, and embodied AI. However, existing methods face several challenges. Learning-based approaches depend on small-scale indoor datasets, limiting the scene diversity and layout complexity. While large language models (LLMs) can leverage diverse text-domain knowledge, they struggle with spatial realism, often producing unnatural object placements that fail to respect common sense. Our key insight is that vision perception can bridge this gap by providing realistic spatial guidance that LLMs lack. To this end, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.02836","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-05T17:59:58Z","cross_cats_sorted":[],"title_canon_sha256":"cd980b143ec53f564fbe3a78d80ed4e7e14d996d78e32b5cd319508b49d77604","abstract_canon_sha256":"a9b49a4c0422a2abd4a117e711bfb6931e84e3cb9461173b79a0bb9fdc410e12"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:58:44.050702Z","signature_b64":"+mlbPgemCk/QRTDuDEJWMpaHPgdC8+ybxHOM8RgWkqBUwjDM/d9Gi6Uh4RNdEhvoegIIFDDTlxNIpV5r1APpBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a7937b7f0448d70bf19b015cdc17dac64d589ad3a7d8e89bb0d57dcc1d8302d8","last_reissued_at":"2026-07-05T10:58:44.050212Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:58:44.050212Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scenethesis: A Language and Vision Agentic Framework for 3D Scene Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aniket Bera, Chen-Hsuan Lin, Lu Ling, Ming-Yu Liu, Tsung-Yi Lin, Yichen Sheng, Yifan Ding, Yunhao Ge, Yu Zeng, Zhaoshuo Li","submitted_at":"2025-05-05T17:59:58Z","abstract_excerpt":"Synthesizing interactive 3D scenes from text is essential for gaming, virtual reality, and embodied AI. However, existing methods face several challenges. Learning-based approaches depend on small-scale indoor datasets, limiting the scene diversity and layout complexity. While large language models (LLMs) can leverage diverse text-domain knowledge, they struggle with spatial realism, often producing unnatural object placements that fail to respect common sense. Our key insight is that vision perception can bridge this gap by providing realistic spatial guidance that LLMs lack. To this end, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.02836","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.02836/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.02836","created_at":"2026-07-05T10:58:44.050272+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.02836v1","created_at":"2026-07-05T10:58:44.050272+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.02836","created_at":"2026-07-05T10:58:44.050272+00:00"},{"alias_kind":"pith_short_12","alias_value":"U6JXW7YEJDLQ","created_at":"2026-07-05T10:58:44.050272+00:00"},{"alias_kind":"pith_short_16","alias_value":"U6JXW7YEJDLQX4M3","created_at":"2026-07-05T10:58:44.050272+00:00"},{"alias_kind":"pith_short_8","alias_value":"U6JXW7YE","created_at":"2026-07-05T10:58:44.050272+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21596","citing_title":"$\\phi$-Scene: Physically Grounded Image-to-3D Scene Reconstruction","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08402","citing_title":"SceneConductor: 3D Scene Generation from a Single Image with Multi-Agent Orchestration","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03994","citing_title":"SimuScene: Simulation-Ready Compositional 3D Scene Reconstruction from a Single Image","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31388","citing_title":"One Video, One World: Turning Monocular Video into Physical 4D Scenes","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29395","citing_title":"NaLA: A 3D Native LLM Layout Agent for High-quality 3D Scene Generation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25326","citing_title":"Perceive-then-Plan: Layout-as-Policy for Monocular 3D Scene Layout Estimation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2601.08454","citing_title":"Real2Sim via Active Perception with Behavior Trees Automatically Generated by VLMs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20837","citing_title":"ArchSIBench: Benchmarking the Architectural Spatial Intelligence of Vision-Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2601.11109","citing_title":"Vision-as-Inverse-Graphics Agent via Interleaved Multimodal Reasoning","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19907","citing_title":"SceneOrchestra: Efficient Agentic 3D Scene Synthesis via Full Tool-Call Trajectory Generation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10578","citing_title":"Rein3D: Reinforced 3D Indoor Scene Generation with Panoramic Video Diffusion Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13035","citing_title":"SceneCritic: A Symbolic Evaluator for 3D Indoor Scene Synthesis","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13800","citing_title":"EmbodiedClaw: Conversational Workflow Execution for Embodied AI Development","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16552","citing_title":"Co-generation of Layout and Shape from Text via Autoregressive 3D Diffusion","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16299","citing_title":"Repurposing 3D Generative Model for Autoregressive Layout Generation","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U6JXW7YEJDLQX4M3AFONYF62YZ","json":"https://pith.science/pith/U6JXW7YEJDLQX4M3AFONYF62YZ.json","graph_json":"https://pith.science/api/pith-number/U6JXW7YEJDLQX4M3AFONYF62YZ/graph.json","events_json":"https://pith.science/api/pith-number/U6JXW7YEJDLQX4M3AFONYF62YZ/events.json","paper":"https://pith.science/paper/U6JXW7YE"},"agent_actions":{"view_html":"https://pith.science/pith/U6JXW7YEJDLQX4M3AFONYF62YZ","download_json":"https://pith.science/pith/U6JXW7YEJDLQX4M3AFONYF62YZ.json","view_paper":"https://pith.science/paper/U6JXW7YE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.02836&json=true","fetch_graph":"https://pith.science/api/pith-number/U6JXW7YEJDLQX4M3AFONYF62YZ/graph.json","fetch_events":"https://pith.science/api/pith-number/U6JXW7YEJDLQX4M3AFONYF62YZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U6JXW7YEJDLQX4M3AFONYF62YZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U6JXW7YEJDLQX4M3AFONYF62YZ/action/storage_attestation","attest_author":"https://pith.science/pith/U6JXW7YEJDLQX4M3AFONYF62YZ/action/author_attestation","sign_citation":"https://pith.science/pith/U6JXW7YEJDLQX4M3AFONYF62YZ/action/citation_signature","submit_replication":"https://pith.science/pith/U6JXW7YEJDLQX4M3AFONYF62YZ/action/replication_record"}},"created_at":"2026-07-05T10:58:44.050272+00:00","updated_at":"2026-07-05T10:58:44.050272+00:00"}