{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:P7QR6XRL6DJRNVAZR6VKOZEPX7","short_pith_number":"pith:P7QR6XRL","schema_version":"1.0","canonical_sha256":"7fe11f5e2bf0d316d4198faaa7648fbfc223715908245a960e3b1450738a46c8","source":{"kind":"arxiv","id":"2405.13218","version":2},"attestation_state":"computed","paper":{"title":"Computational Tradeoffs in Image Synthesis: Diffusion, Masked-Token, and Next-Token Prediction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Luke Zettlemoyer, Maciej Kilian, Varun Jampani","submitted_at":"2024-05-21T21:49:39Z","abstract_excerpt":"Nearly every recent image synthesis approach, including diffusion, masked-token prediction, and next-token prediction, uses a Transformer network architecture. Despite this common backbone, there has been no direct, compute controlled comparison of how these approaches affect performance and efficiency. We analyze the scalability of each approach through the lens of compute budget measured in FLOPs. We find that token prediction methods, led by next-token prediction, significantly outperform diffusion on prompt following. On image quality, while next-token prediction initially performs better,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.13218","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-21T21:49:39Z","cross_cats_sorted":[],"title_canon_sha256":"58300a264860bd0ea5c58b13c59d6020744b78af21b7eb2885a7c6147833a7f7","abstract_canon_sha256":"c12b3e49d51ee9265222f58616295c2318a9b9a152c00280d5013d175e4c5466"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:41.027997Z","signature_b64":"nfTME+WHlXO1ECP9m54I+MOJGuAfb1UUg0Oot8b5zbhniY8pkgaRLrokcVbi5pTolJb5nMamTMmZTZjvmT8gCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7fe11f5e2bf0d316d4198faaa7648fbfc223715908245a960e3b1450738a46c8","last_reissued_at":"2026-07-05T08:22:41.027532Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:41.027532Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Computational Tradeoffs in Image Synthesis: Diffusion, Masked-Token, and Next-Token Prediction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Luke Zettlemoyer, Maciej Kilian, Varun Jampani","submitted_at":"2024-05-21T21:49:39Z","abstract_excerpt":"Nearly every recent image synthesis approach, including diffusion, masked-token prediction, and next-token prediction, uses a Transformer network architecture. Despite this common backbone, there has been no direct, compute controlled comparison of how these approaches affect performance and efficiency. We analyze the scalability of each approach through the lens of compute budget measured in FLOPs. We find that token prediction methods, led by next-token prediction, significantly outperform diffusion on prompt following. On image quality, while next-token prediction initially performs better,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.13218","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.13218/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.13218","created_at":"2026-07-05T08:22:41.027588+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.13218v2","created_at":"2026-07-05T08:22:41.027588+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.13218","created_at":"2026-07-05T08:22:41.027588+00:00"},{"alias_kind":"pith_short_12","alias_value":"P7QR6XRL6DJR","created_at":"2026-07-05T08:22:41.027588+00:00"},{"alias_kind":"pith_short_16","alias_value":"P7QR6XRL6DJRNVAZ","created_at":"2026-07-05T08:22:41.027588+00:00"},{"alias_kind":"pith_short_8","alias_value":"P7QR6XRL","created_at":"2026-07-05T08:22:41.027588+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26230","citing_title":"Geometry-Aware Representation Denoising for Robust Multi-view 3D Reconstruction","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2412.03603","citing_title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12013","citing_title":"L2P: Unlocking Latent Potential for Pixel Generation","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P7QR6XRL6DJRNVAZR6VKOZEPX7","json":"https://pith.science/pith/P7QR6XRL6DJRNVAZR6VKOZEPX7.json","graph_json":"https://pith.science/api/pith-number/P7QR6XRL6DJRNVAZR6VKOZEPX7/graph.json","events_json":"https://pith.science/api/pith-number/P7QR6XRL6DJRNVAZR6VKOZEPX7/events.json","paper":"https://pith.science/paper/P7QR6XRL"},"agent_actions":{"view_html":"https://pith.science/pith/P7QR6XRL6DJRNVAZR6VKOZEPX7","download_json":"https://pith.science/pith/P7QR6XRL6DJRNVAZR6VKOZEPX7.json","view_paper":"https://pith.science/paper/P7QR6XRL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.13218&json=true","fetch_graph":"https://pith.science/api/pith-number/P7QR6XRL6DJRNVAZR6VKOZEPX7/graph.json","fetch_events":"https://pith.science/api/pith-number/P7QR6XRL6DJRNVAZR6VKOZEPX7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P7QR6XRL6DJRNVAZR6VKOZEPX7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P7QR6XRL6DJRNVAZR6VKOZEPX7/action/storage_attestation","attest_author":"https://pith.science/pith/P7QR6XRL6DJRNVAZR6VKOZEPX7/action/author_attestation","sign_citation":"https://pith.science/pith/P7QR6XRL6DJRNVAZR6VKOZEPX7/action/citation_signature","submit_replication":"https://pith.science/pith/P7QR6XRL6DJRNVAZR6VKOZEPX7/action/replication_record"}},"created_at":"2026-07-05T08:22:41.027588+00:00","updated_at":"2026-07-05T08:22:41.027588+00:00"}