{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:X3VGS54BNS7EQXVEXM46V6IG6T","short_pith_number":"pith:X3VGS54B","schema_version":"1.0","canonical_sha256":"beea6977816cbe485ea4bb39eaf906f4c77cc94a2f9960ef20ec83cbf624728c","source":{"kind":"arxiv","id":"2410.02705","version":3},"attestation_state":"computed","paper":{"title":"ControlAR: Controllable Image Generation with Autoregressive Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haocheng Shen, Longjin Ran, Peize Sun, Shoufa Chen, Tianheng Cheng, Wenyu Liu, Xiaoxin Chen, Xinggang Wang, Zongming Li","submitted_at":"2024-10-03T17:28:07Z","abstract_excerpt":"Autoregressive (AR) models have reformulated image generation as next-token prediction, demonstrating remarkable potential and emerging as strong competitors to diffusion models. However, control-to-image generation, akin to ControlNet, remains largely unexplored within AR models. Although a natural approach, inspired by advancements in Large Language Models, is to tokenize control images into tokens and prefill them into the autoregressive model before decoding image tokens, it still falls short in generation quality compared to ControlNet and suffers from inefficiency. To this end, we introd"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.02705","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-03T17:28:07Z","cross_cats_sorted":[],"title_canon_sha256":"cb58b10f1d4ba8d4097d8cb71f0ffa3f4f6c49b801abdefcd62dd042b2899c62","abstract_canon_sha256":"733b7b659ab493819b1616413fab8739f8ce9c75cea7eaa05b3e17440e94591b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:27:16.320272Z","signature_b64":"s4n/hPG2qYuJESSZjz80J7KTqJ3tvDmJ2voA4mFN4CQ3HM0W6tU+3etYKrkmqp3JngZHyZPupTIkl15KatVqAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"beea6977816cbe485ea4bb39eaf906f4c77cc94a2f9960ef20ec83cbf624728c","last_reissued_at":"2026-07-05T10:27:16.319764Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:27:16.319764Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ControlAR: Controllable Image Generation with Autoregressive Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haocheng Shen, Longjin Ran, Peize Sun, Shoufa Chen, Tianheng Cheng, Wenyu Liu, Xiaoxin Chen, Xinggang Wang, Zongming Li","submitted_at":"2024-10-03T17:28:07Z","abstract_excerpt":"Autoregressive (AR) models have reformulated image generation as next-token prediction, demonstrating remarkable potential and emerging as strong competitors to diffusion models. However, control-to-image generation, akin to ControlNet, remains largely unexplored within AR models. Although a natural approach, inspired by advancements in Large Language Models, is to tokenize control images into tokens and prefill them into the autoregressive model before decoding image tokens, it still falls short in generation quality compared to ControlNet and suffers from inefficiency. To this end, we introd"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02705","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02705/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.02705","created_at":"2026-07-05T10:27:16.319825+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.02705v3","created_at":"2026-07-05T10:27:16.319825+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02705","created_at":"2026-07-05T10:27:16.319825+00:00"},{"alias_kind":"pith_short_12","alias_value":"X3VGS54BNS7E","created_at":"2026-07-05T10:27:16.319825+00:00"},{"alias_kind":"pith_short_16","alias_value":"X3VGS54BNS7EQXVE","created_at":"2026-07-05T10:27:16.319825+00:00"},{"alias_kind":"pith_short_8","alias_value":"X3VGS54B","created_at":"2026-07-05T10:27:16.319825+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.22394","citing_title":"PacTure: Efficient PBR Texture Generation on Packed Views with Visual Autoregressive Models","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2603.00166","citing_title":"Exploring the AI Obedience: Why is Generating a Pure Color Image Harder than CyberPunk?","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12138","citing_title":"Design Your Ad: Personalized Advertising Image and Text Generation with Unified Autoregressive Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21450","citing_title":"VARestorer: One-Step VAR Distillation for Real-World Image Super-Resolution","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X3VGS54BNS7EQXVEXM46V6IG6T","json":"https://pith.science/pith/X3VGS54BNS7EQXVEXM46V6IG6T.json","graph_json":"https://pith.science/api/pith-number/X3VGS54BNS7EQXVEXM46V6IG6T/graph.json","events_json":"https://pith.science/api/pith-number/X3VGS54BNS7EQXVEXM46V6IG6T/events.json","paper":"https://pith.science/paper/X3VGS54B"},"agent_actions":{"view_html":"https://pith.science/pith/X3VGS54BNS7EQXVEXM46V6IG6T","download_json":"https://pith.science/pith/X3VGS54BNS7EQXVEXM46V6IG6T.json","view_paper":"https://pith.science/paper/X3VGS54B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.02705&json=true","fetch_graph":"https://pith.science/api/pith-number/X3VGS54BNS7EQXVEXM46V6IG6T/graph.json","fetch_events":"https://pith.science/api/pith-number/X3VGS54BNS7EQXVEXM46V6IG6T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X3VGS54BNS7EQXVEXM46V6IG6T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X3VGS54BNS7EQXVEXM46V6IG6T/action/storage_attestation","attest_author":"https://pith.science/pith/X3VGS54BNS7EQXVEXM46V6IG6T/action/author_attestation","sign_citation":"https://pith.science/pith/X3VGS54BNS7EQXVEXM46V6IG6T/action/citation_signature","submit_replication":"https://pith.science/pith/X3VGS54BNS7EQXVEXM46V6IG6T/action/replication_record"}},"created_at":"2026-07-05T10:27:16.319825+00:00","updated_at":"2026-07-05T10:27:16.319825+00:00"}