{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:WGMS5PB4VMZ4PHQLWU6RHICYQP","short_pith_number":"pith:WGMS5PB4","schema_version":"1.0","canonical_sha256":"b1992ebc3cab33c79e0bb53d13a05883d34bf27777050eda121a1454c76f9dab","source":{"kind":"arxiv","id":"2011.10650","version":2},"attestation_state":"computed","paper":{"title":"Very Deep VAEs Generalize Autoregressive Models and Can Outperform Them on Images","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Rewon Child","submitted_at":"2020-11-20T21:35:31Z","abstract_excerpt":"We present a hierarchical VAE that, for the first time, generates samples quickly while outperforming the PixelCNN in log-likelihood on all natural image benchmarks. We begin by observing that, in theory, VAEs can actually represent autoregressive models, as well as faster, better models if they exist, when made sufficiently deep. Despite this, autoregressive models have historically outperformed VAEs in log-likelihood. We test if insufficient depth explains why by scaling a VAE to greater stochastic depth than previously explored and evaluating it CIFAR-10, ImageNet, and FFHQ. In comparison t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2011.10650","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-11-20T21:35:31Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"6ddfabe38b7fd49435b08f4bc9a642f6fa5eb577a1be7eeebffc24f4fc5ffc3b","abstract_canon_sha256":"1b2c9b6b1a2445d63319424c6517bf1ca16c5a29e47ad6f67ed36b8b08a548a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:23:59.720648Z","signature_b64":"eC3mEb6Mzl3QPu0CFYshlt6a54Y6JpvG9s+VHGEb1/nchYd9lCP+tDiU0GWdVkpKLJzrpzZ7p38HK5SXYkfzAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b1992ebc3cab33c79e0bb53d13a05883d34bf27777050eda121a1454c76f9dab","last_reissued_at":"2026-07-05T02:23:59.720082Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:23:59.720082Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Very Deep VAEs Generalize Autoregressive Models and Can Outperform Them on Images","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Rewon Child","submitted_at":"2020-11-20T21:35:31Z","abstract_excerpt":"We present a hierarchical VAE that, for the first time, generates samples quickly while outperforming the PixelCNN in log-likelihood on all natural image benchmarks. We begin by observing that, in theory, VAEs can actually represent autoregressive models, as well as faster, better models if they exist, when made sufficiently deep. Despite this, autoregressive models have historically outperformed VAEs in log-likelihood. We test if insufficient depth explains why by scaling a VAE to greater stochastic depth than previously explored and evaluating it CIFAR-10, ImageNet, and FFHQ. In comparison t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.10650","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.10650/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2011.10650","created_at":"2026-07-05T02:23:59.720145+00:00"},{"alias_kind":"arxiv_version","alias_value":"2011.10650v2","created_at":"2026-07-05T02:23:59.720145+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.10650","created_at":"2026-07-05T02:23:59.720145+00:00"},{"alias_kind":"pith_short_12","alias_value":"WGMS5PB4VMZ4","created_at":"2026-07-05T02:23:59.720145+00:00"},{"alias_kind":"pith_short_16","alias_value":"WGMS5PB4VMZ4PHQL","created_at":"2026-07-05T02:23:59.720145+00:00"},{"alias_kind":"pith_short_8","alias_value":"WGMS5PB4","created_at":"2026-07-05T02:23:59.720145+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08198","citing_title":"Unpaired Joint Distribution Modeling via Multi-Scale Image Representations","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22377","citing_title":"Multigrid Training for Molecular Generation using Graph Neural Networks","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31363","citing_title":"Language-Assisted Super-Resolution from Real-World Low-Resolution Patches","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31363","citing_title":"Language-Assisted Super-Resolution from Real-World Low-Resolution Patches","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22851","citing_title":"VAMP-Diff: VampPrior Latent Diffusion for Photoplethysmography Modeling","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2308.08089","citing_title":"DragNUWA: Fine-grained Control in Video Generation by Integrating Text, Image, and Trajectory","ref_index":150,"is_internal_anchor":false},{"citing_arxiv_id":"2101.02388","citing_title":"Knowledge Distillation in Iterative Generative Models for Improved Sampling Speed","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2102.09672","citing_title":"Improved Denoising Diffusion Probabilistic Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2104.10157","citing_title":"VideoGPT: Video Generation using VQ-VAE and Transformers","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2105.05233","citing_title":"Diffusion Models Beat GANs on Image Synthesis","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2112.10752","citing_title":"High-Resolution Image Synthesis with Latent Diffusion Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2301.04104","citing_title":"Mastering Diverse Domains through World Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2204.06125","citing_title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WGMS5PB4VMZ4PHQLWU6RHICYQP","json":"https://pith.science/pith/WGMS5PB4VMZ4PHQLWU6RHICYQP.json","graph_json":"https://pith.science/api/pith-number/WGMS5PB4VMZ4PHQLWU6RHICYQP/graph.json","events_json":"https://pith.science/api/pith-number/WGMS5PB4VMZ4PHQLWU6RHICYQP/events.json","paper":"https://pith.science/paper/WGMS5PB4"},"agent_actions":{"view_html":"https://pith.science/pith/WGMS5PB4VMZ4PHQLWU6RHICYQP","download_json":"https://pith.science/pith/WGMS5PB4VMZ4PHQLWU6RHICYQP.json","view_paper":"https://pith.science/paper/WGMS5PB4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2011.10650&json=true","fetch_graph":"https://pith.science/api/pith-number/WGMS5PB4VMZ4PHQLWU6RHICYQP/graph.json","fetch_events":"https://pith.science/api/pith-number/WGMS5PB4VMZ4PHQLWU6RHICYQP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WGMS5PB4VMZ4PHQLWU6RHICYQP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WGMS5PB4VMZ4PHQLWU6RHICYQP/action/storage_attestation","attest_author":"https://pith.science/pith/WGMS5PB4VMZ4PHQLWU6RHICYQP/action/author_attestation","sign_citation":"https://pith.science/pith/WGMS5PB4VMZ4PHQLWU6RHICYQP/action/citation_signature","submit_replication":"https://pith.science/pith/WGMS5PB4VMZ4PHQLWU6RHICYQP/action/replication_record"}},"created_at":"2026-07-05T02:23:59.720145+00:00","updated_at":"2026-07-05T02:23:59.720145+00:00"}