{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NU5KHMMYPGTGO3ADYEF5IFW6JZ","short_pith_number":"pith:NU5KHMMY","schema_version":"1.0","canonical_sha256":"6d3aa3b19879a6676c03c10bd416de4e5325785818df877d22c7f0c494d2da6e","source":{"kind":"arxiv","id":"2309.07986","version":2},"attestation_state":"computed","paper":{"title":"Viewpoint Textual Inversion: Discovering Scene Representations and 3D View Control in 2D Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"James Burgess, Kuan-Chieh Wang, Serena Yeung-Levy","submitted_at":"2023-09-14T18:52:16Z","abstract_excerpt":"Text-to-image diffusion models generate impressive and realistic images, but do they learn to represent the 3D world from only 2D supervision? We demonstrate that yes, certain 3D scene representations are encoded in the text embedding space of models like Stable Diffusion. Our approach, Viewpoint Neural Textual Inversion (ViewNeTI), is to discover 3D view tokens; these tokens control the 3D viewpoint - the rendering pose in a scene - of generated images. Specifically, we train a small neural mapper to take continuous camera viewpoint parameters and predict a view token (a word embedding). This"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.07986","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-09-14T18:52:16Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"a5b528b56785c07fa71940dde01631fdd11a369bfc3f1f5c82401c47c2fcd093","abstract_canon_sha256":"3fe2a3b9e0a90cfb3e7779e5ed595c8d6e0c5f056892fe34a19727d74be1eb21"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:48:39.312148Z","signature_b64":"eUliFTyJweDHee4iVbNH2z9FH9KoEMvYaPkZe+xEJpxIGFrpIvRCC39NhzoiQLQTUOqRSEBcQALOMOOwD5UeDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6d3aa3b19879a6676c03c10bd416de4e5325785818df877d22c7f0c494d2da6e","last_reissued_at":"2026-07-05T08:48:39.311692Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:48:39.311692Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Viewpoint Textual Inversion: Discovering Scene Representations and 3D View Control in 2D Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"James Burgess, Kuan-Chieh Wang, Serena Yeung-Levy","submitted_at":"2023-09-14T18:52:16Z","abstract_excerpt":"Text-to-image diffusion models generate impressive and realistic images, but do they learn to represent the 3D world from only 2D supervision? We demonstrate that yes, certain 3D scene representations are encoded in the text embedding space of models like Stable Diffusion. Our approach, Viewpoint Neural Textual Inversion (ViewNeTI), is to discover 3D view tokens; these tokens control the 3D viewpoint - the rendering pose in a scene - of generated images. Specifically, we train a small neural mapper to take continuous camera viewpoint parameters and predict a view token (a word embedding). This"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.07986","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.07986/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.07986","created_at":"2026-07-05T08:48:39.311752+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.07986v2","created_at":"2026-07-05T08:48:39.311752+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.07986","created_at":"2026-07-05T08:48:39.311752+00:00"},{"alias_kind":"pith_short_12","alias_value":"NU5KHMMYPGTG","created_at":"2026-07-05T08:48:39.311752+00:00"},{"alias_kind":"pith_short_16","alias_value":"NU5KHMMYPGTGO3AD","created_at":"2026-07-05T08:48:39.311752+00:00"},{"alias_kind":"pith_short_8","alias_value":"NU5KHMMY","created_at":"2026-07-05T08:48:39.311752+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.02690","citing_title":"FoundHand: Large-Scale Domain-Specific Learning for Controllable Hand Image Generation","ref_index":6,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NU5KHMMYPGTGO3ADYEF5IFW6JZ","json":"https://pith.science/pith/NU5KHMMYPGTGO3ADYEF5IFW6JZ.json","graph_json":"https://pith.science/api/pith-number/NU5KHMMYPGTGO3ADYEF5IFW6JZ/graph.json","events_json":"https://pith.science/api/pith-number/NU5KHMMYPGTGO3ADYEF5IFW6JZ/events.json","paper":"https://pith.science/paper/NU5KHMMY"},"agent_actions":{"view_html":"https://pith.science/pith/NU5KHMMYPGTGO3ADYEF5IFW6JZ","download_json":"https://pith.science/pith/NU5KHMMYPGTGO3ADYEF5IFW6JZ.json","view_paper":"https://pith.science/paper/NU5KHMMY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.07986&json=true","fetch_graph":"https://pith.science/api/pith-number/NU5KHMMYPGTGO3ADYEF5IFW6JZ/graph.json","fetch_events":"https://pith.science/api/pith-number/NU5KHMMYPGTGO3ADYEF5IFW6JZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NU5KHMMYPGTGO3ADYEF5IFW6JZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NU5KHMMYPGTGO3ADYEF5IFW6JZ/action/storage_attestation","attest_author":"https://pith.science/pith/NU5KHMMYPGTGO3ADYEF5IFW6JZ/action/author_attestation","sign_citation":"https://pith.science/pith/NU5KHMMYPGTGO3ADYEF5IFW6JZ/action/citation_signature","submit_replication":"https://pith.science/pith/NU5KHMMYPGTGO3ADYEF5IFW6JZ/action/replication_record"}},"created_at":"2026-07-05T08:48:39.311752+00:00","updated_at":"2026-07-05T08:48:39.311752+00:00"}