{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:TZJPDRMWR4LOXHI7MZNVUGIS6N","short_pith_number":"pith:TZJPDRMW","schema_version":"1.0","canonical_sha256":"9e52f1c5968f16eb9d1f665b5a1912f36631a97611d3976849684500c3fdbd83","source":{"kind":"arxiv","id":"2105.06458","version":1},"attestation_state":"computed","paper":{"title":"High-Resolution Complex Scene Synthesis with Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bj\\\"orn Ommer, Manuel Jahn, Robin Rombach","submitted_at":"2021-05-13T17:56:07Z","abstract_excerpt":"The use of coarse-grained layouts for controllable synthesis of complex scene images via deep generative models has recently gained popularity. However, results of current approaches still fall short of their promise of high-resolution synthesis. We hypothesize that this is mostly due to the highly engineered nature of these approaches which often rely on auxiliary losses and intermediate steps such as mask generators. In this note, we present an orthogonal approach to this task, where the generative model is based on pure likelihood training without additional objectives. To do so, we first o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.06458","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-05-13T17:56:07Z","cross_cats_sorted":[],"title_canon_sha256":"5c2f2edac0a72d6371e94ec9374aa582c34bc7ef183af0cdb9b582891b740fc2","abstract_canon_sha256":"4c96e998b1bc726d7d0eb6143a659a3ccd6e286328f77fda499cc7a90dce787a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:40:07.889215Z","signature_b64":"3bioyDp2A7WixpUSdt8ldQIwion1GITne1O8MFF06Vo1U2PguDJW2wFtqzpOSLCwro/oUSNMWQ1OzXqkJ3XhCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e52f1c5968f16eb9d1f665b5a1912f36631a97611d3976849684500c3fdbd83","last_reissued_at":"2026-07-05T02:40:07.888737Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:40:07.888737Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"High-Resolution Complex Scene Synthesis with Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bj\\\"orn Ommer, Manuel Jahn, Robin Rombach","submitted_at":"2021-05-13T17:56:07Z","abstract_excerpt":"The use of coarse-grained layouts for controllable synthesis of complex scene images via deep generative models has recently gained popularity. However, results of current approaches still fall short of their promise of high-resolution synthesis. We hypothesize that this is mostly due to the highly engineered nature of these approaches which often rely on auxiliary losses and intermediate steps such as mask generators. In this note, we present an orthogonal approach to this task, where the generative model is based on pure likelihood training without additional objectives. To do so, we first o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.06458","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.06458/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.06458","created_at":"2026-07-05T02:40:07.888795+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.06458v1","created_at":"2026-07-05T02:40:07.888795+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.06458","created_at":"2026-07-05T02:40:07.888795+00:00"},{"alias_kind":"pith_short_12","alias_value":"TZJPDRMWR4LO","created_at":"2026-07-05T02:40:07.888795+00:00"},{"alias_kind":"pith_short_16","alias_value":"TZJPDRMWR4LOXHI7","created_at":"2026-07-05T02:40:07.888795+00:00"},{"alias_kind":"pith_short_8","alias_value":"TZJPDRMW","created_at":"2026-07-05T02:40:07.888795+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2112.10752","citing_title":"High-Resolution Image Synthesis with Latent Diffusion Models","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TZJPDRMWR4LOXHI7MZNVUGIS6N","json":"https://pith.science/pith/TZJPDRMWR4LOXHI7MZNVUGIS6N.json","graph_json":"https://pith.science/api/pith-number/TZJPDRMWR4LOXHI7MZNVUGIS6N/graph.json","events_json":"https://pith.science/api/pith-number/TZJPDRMWR4LOXHI7MZNVUGIS6N/events.json","paper":"https://pith.science/paper/TZJPDRMW"},"agent_actions":{"view_html":"https://pith.science/pith/TZJPDRMWR4LOXHI7MZNVUGIS6N","download_json":"https://pith.science/pith/TZJPDRMWR4LOXHI7MZNVUGIS6N.json","view_paper":"https://pith.science/paper/TZJPDRMW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.06458&json=true","fetch_graph":"https://pith.science/api/pith-number/TZJPDRMWR4LOXHI7MZNVUGIS6N/graph.json","fetch_events":"https://pith.science/api/pith-number/TZJPDRMWR4LOXHI7MZNVUGIS6N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TZJPDRMWR4LOXHI7MZNVUGIS6N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TZJPDRMWR4LOXHI7MZNVUGIS6N/action/storage_attestation","attest_author":"https://pith.science/pith/TZJPDRMWR4LOXHI7MZNVUGIS6N/action/author_attestation","sign_citation":"https://pith.science/pith/TZJPDRMWR4LOXHI7MZNVUGIS6N/action/citation_signature","submit_replication":"https://pith.science/pith/TZJPDRMWR4LOXHI7MZNVUGIS6N/action/replication_record"}},"created_at":"2026-07-05T02:40:07.888795+00:00","updated_at":"2026-07-05T02:40:07.888795+00:00"}