{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:PEOZZFULK6PF4MA44DSBYB5VCB","short_pith_number":"pith:PEOZZFUL","schema_version":"1.0","canonical_sha256":"791d9c968b579e5e301ce0e41c07b510424b13b3491baa930fa386137b700f93","source":{"kind":"arxiv","id":"2111.13152","version":3},"attestation_state":"computed","paper":{"title":"Scene Representation Transformer: Geometry-Free Novel View Synthesis Through Set-Latent Scene Representations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GR","cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Alexey Dosovitskiy, Andrea Tagliasacchi, Daniel Duckworth, Etienne Pot, Henning Meyer, Jakob Uszkoreit, Klaus Greff, Mario Lucic, Mehdi S. M. Sajjadi, Noha Radwan, Suhani Vora, Thomas Funkhouser, Urs Bergmann","submitted_at":"2021-11-25T16:18:56Z","abstract_excerpt":"A classical problem in computer vision is to infer a 3D scene representation from few images that can be used to render novel views at interactive rates. Previous work focuses on reconstructing pre-defined 3D representations, e.g. textured meshes, or implicit representations, e.g. radiance fields, and often requires input images with precise camera poses and long processing times for each novel scene.\n  In this work, we propose the Scene Representation Transformer (SRT), a method which processes posed or unposed RGB images of a new area, infers a \"set-latent scene representation\", and synthesi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.13152","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-11-25T16:18:56Z","cross_cats_sorted":["cs.AI","cs.GR","cs.LG","cs.RO"],"title_canon_sha256":"8644822b08fbc037d5932fb3454efdc0373c8c8d71968e60d4fc42b63f903c75","abstract_canon_sha256":"386fa1e190ca580c1cf4d9ffaca41a04c870e7c1b6d86653def46f9204e61069"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:09:16.547118Z","signature_b64":"W7Xb+d+J9x/NEEA7uHLpdD1Ue87xRg740zroV6vbq0HA6azawjq/vX6vs/k4LHfshYLd3Leh5j/Jti3QF6J3DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"791d9c968b579e5e301ce0e41c07b510424b13b3491baa930fa386137b700f93","last_reissued_at":"2026-07-05T04:09:16.546569Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:09:16.546569Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scene Representation Transformer: Geometry-Free Novel View Synthesis Through Set-Latent Scene Representations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GR","cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Alexey Dosovitskiy, Andrea Tagliasacchi, Daniel Duckworth, Etienne Pot, Henning Meyer, Jakob Uszkoreit, Klaus Greff, Mario Lucic, Mehdi S. M. Sajjadi, Noha Radwan, Suhani Vora, Thomas Funkhouser, Urs Bergmann","submitted_at":"2021-11-25T16:18:56Z","abstract_excerpt":"A classical problem in computer vision is to infer a 3D scene representation from few images that can be used to render novel views at interactive rates. Previous work focuses on reconstructing pre-defined 3D representations, e.g. textured meshes, or implicit representations, e.g. radiance fields, and often requires input images with precise camera poses and long processing times for each novel scene.\n  In this work, we propose the Scene Representation Transformer (SRT), a method which processes posed or unposed RGB images of a new area, infers a \"set-latent scene representation\", and synthesi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.13152","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.13152/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.13152","created_at":"2026-07-05T04:09:16.546633+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.13152v3","created_at":"2026-07-05T04:09:16.546633+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.13152","created_at":"2026-07-05T04:09:16.546633+00:00"},{"alias_kind":"pith_short_12","alias_value":"PEOZZFULK6PF","created_at":"2026-07-05T04:09:16.546633+00:00"},{"alias_kind":"pith_short_16","alias_value":"PEOZZFULK6PF4MA4","created_at":"2026-07-05T04:09:16.546633+00:00"},{"alias_kind":"pith_short_8","alias_value":"PEOZZFUL","created_at":"2026-07-05T04:09:16.546633+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12938","citing_title":"CRePE: Curved Ray Expectation Positional Encoding for Unified-Camera-Controlled Video Generation","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PEOZZFULK6PF4MA44DSBYB5VCB","json":"https://pith.science/pith/PEOZZFULK6PF4MA44DSBYB5VCB.json","graph_json":"https://pith.science/api/pith-number/PEOZZFULK6PF4MA44DSBYB5VCB/graph.json","events_json":"https://pith.science/api/pith-number/PEOZZFULK6PF4MA44DSBYB5VCB/events.json","paper":"https://pith.science/paper/PEOZZFUL"},"agent_actions":{"view_html":"https://pith.science/pith/PEOZZFULK6PF4MA44DSBYB5VCB","download_json":"https://pith.science/pith/PEOZZFULK6PF4MA44DSBYB5VCB.json","view_paper":"https://pith.science/paper/PEOZZFUL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.13152&json=true","fetch_graph":"https://pith.science/api/pith-number/PEOZZFULK6PF4MA44DSBYB5VCB/graph.json","fetch_events":"https://pith.science/api/pith-number/PEOZZFULK6PF4MA44DSBYB5VCB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PEOZZFULK6PF4MA44DSBYB5VCB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PEOZZFULK6PF4MA44DSBYB5VCB/action/storage_attestation","attest_author":"https://pith.science/pith/PEOZZFULK6PF4MA44DSBYB5VCB/action/author_attestation","sign_citation":"https://pith.science/pith/PEOZZFULK6PF4MA44DSBYB5VCB/action/citation_signature","submit_replication":"https://pith.science/pith/PEOZZFULK6PF4MA44DSBYB5VCB/action/replication_record"}},"created_at":"2026-07-05T04:09:16.546633+00:00","updated_at":"2026-07-05T04:09:16.546633+00:00"}