{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:57S5RMXPEO3GHQGE3LCAT4Y5RN","short_pith_number":"pith:57S5RMXP","schema_version":"1.0","canonical_sha256":"efe5d8b2ef23b663c0c4dac409f31d8b59492cbc047e2635494225875cf44279","source":{"kind":"arxiv","id":"2410.16770","version":2},"attestation_state":"computed","paper":{"title":"The Scene Language: Representing Scenes with Programs, Words, and Embeddings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jiajun Wu, Matt Zhou, Shangzhe Wu, Yunzhi Zhang, Zizhang Li","submitted_at":"2024-10-22T07:40:20Z","abstract_excerpt":"We introduce the Scene Language, a visual scene representation that concisely and precisely describes the structure, semantics, and identity of visual scenes. It represents a scene with three key components: a program that specifies the hierarchical and relational structure of entities in the scene, words in natural language that summarize the semantic class of each entity, and embeddings that capture the visual identity of each entity. This representation can be inferred from pre-trained language models via a training-free inference technique, given text or image inputs. The resulting scene c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.16770","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-22T07:40:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"26dc5af37dd8674640e44a2a1fd35ef6f336755d1d120cf9a9a9f65e83cdaf85","abstract_canon_sha256":"55b85445a1f0fd2af0f84c8325b7a5cd72b9263ddc0baddb9b3f004fb2c767b1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:16.325812Z","signature_b64":"iSyS3ybUijsThraGhDM4a1m5V3m29vOsM+elMJJKxwHX3q+vpQCXF1cIL1XcAb0LG42iyyS5JvlfPlJOTJDlBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"efe5d8b2ef23b663c0c4dac409f31d8b59492cbc047e2635494225875cf44279","last_reissued_at":"2026-07-05T10:41:16.325295Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:16.325295Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Scene Language: Representing Scenes with Programs, Words, and Embeddings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jiajun Wu, Matt Zhou, Shangzhe Wu, Yunzhi Zhang, Zizhang Li","submitted_at":"2024-10-22T07:40:20Z","abstract_excerpt":"We introduce the Scene Language, a visual scene representation that concisely and precisely describes the structure, semantics, and identity of visual scenes. It represents a scene with three key components: a program that specifies the hierarchical and relational structure of entities in the scene, words in natural language that summarize the semantic class of each entity, and embeddings that capture the visual identity of each entity. This representation can be inferred from pre-trained language models via a training-free inference technique, given text or image inputs. The resulting scene c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.16770","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.16770/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.16770","created_at":"2026-07-05T10:41:16.325372+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.16770v2","created_at":"2026-07-05T10:41:16.325372+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.16770","created_at":"2026-07-05T10:41:16.325372+00:00"},{"alias_kind":"pith_short_12","alias_value":"57S5RMXPEO3G","created_at":"2026-07-05T10:41:16.325372+00:00"},{"alias_kind":"pith_short_16","alias_value":"57S5RMXPEO3GHQGE","created_at":"2026-07-05T10:41:16.325372+00:00"},{"alias_kind":"pith_short_8","alias_value":"57S5RMXP","created_at":"2026-07-05T10:41:16.325372+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02407","citing_title":"Text-Driven 3D Indoor Scene Synthesis in Non-Manhattan Environments","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25326","citing_title":"Perceive-then-Plan: Layout-as-Policy for Monocular 3D Scene Layout Estimation","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2601.02078","citing_title":"Genie Sim 3.0 : A High-Fidelity Comprehensive Simulation Platform for Humanoid Robot","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/57S5RMXPEO3GHQGE3LCAT4Y5RN","json":"https://pith.science/pith/57S5RMXPEO3GHQGE3LCAT4Y5RN.json","graph_json":"https://pith.science/api/pith-number/57S5RMXPEO3GHQGE3LCAT4Y5RN/graph.json","events_json":"https://pith.science/api/pith-number/57S5RMXPEO3GHQGE3LCAT4Y5RN/events.json","paper":"https://pith.science/paper/57S5RMXP"},"agent_actions":{"view_html":"https://pith.science/pith/57S5RMXPEO3GHQGE3LCAT4Y5RN","download_json":"https://pith.science/pith/57S5RMXPEO3GHQGE3LCAT4Y5RN.json","view_paper":"https://pith.science/paper/57S5RMXP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.16770&json=true","fetch_graph":"https://pith.science/api/pith-number/57S5RMXPEO3GHQGE3LCAT4Y5RN/graph.json","fetch_events":"https://pith.science/api/pith-number/57S5RMXPEO3GHQGE3LCAT4Y5RN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/57S5RMXPEO3GHQGE3LCAT4Y5RN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/57S5RMXPEO3GHQGE3LCAT4Y5RN/action/storage_attestation","attest_author":"https://pith.science/pith/57S5RMXPEO3GHQGE3LCAT4Y5RN/action/author_attestation","sign_citation":"https://pith.science/pith/57S5RMXPEO3GHQGE3LCAT4Y5RN/action/citation_signature","submit_replication":"https://pith.science/pith/57S5RMXPEO3GHQGE3LCAT4Y5RN/action/replication_record"}},"created_at":"2026-07-05T10:41:16.325372+00:00","updated_at":"2026-07-05T10:41:16.325372+00:00"}