{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DL7TW56R37CEWLKSL4MLDRDJLW","short_pith_number":"pith:DL7TW56R","schema_version":"1.0","canonical_sha256":"1aff3b77d1dfc44b2d525f18b1c4695d9611a89efe769fa3a660e00e4a25e716","source":{"kind":"arxiv","id":"2411.05902","version":2},"attestation_state":"computed","paper":{"title":"Autoregressive Models in Vision: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chaofan Tao, Chengyue Wu, Gongye Liu, Guillermo Sapiro, Hongxia Yang, Huaxiu Yao, Hui Shen, Jiebo Luo, Jinfa Huang, Jing Xiong, Lingpeng Kong, Lun Huang, Mi Zhang, Ngai Wong, Ping Luo, Shen Yan, Taiqiang Wu, Yao Mu, Yuan Yao, Zhongwei Wan","submitted_at":"2024-11-08T17:15:12Z","abstract_excerpt":"Autoregressive modeling has been a huge success in the field of natural language processing (NLP). Recently, autoregressive models have emerged as a significant area of focus in computer vision, where they excel in producing high-quality visual content. Autoregressive models in NLP typically operate on subword tokens. However, the representation strategy in computer vision can vary in different levels, i.e., pixel-level, token-level, or scale-level, reflecting the diverse and hierarchical nature of visual data compared to the sequential structure of language. This survey comprehensively examin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.05902","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-08T17:15:12Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"4951d837abbdd5b50fdb46a24428f9c60a265e5cc769d376f3e3bcf77bc8cc9c","abstract_canon_sha256":"23d1697dc64990907e8d591b5a959a0e1fcb030a72d180a5159a9d0c01200835"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:15.589051Z","signature_b64":"uma7aPGqWHzLwjsxWu08ha2NAgt3XV/3sY93I33jpkh4/CuZsBveCiPsq3+mE0uNNjmAZDmD+htk+phX6o7kBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1aff3b77d1dfc44b2d525f18b1c4695d9611a89efe769fa3a660e00e4a25e716","last_reissued_at":"2026-07-05T11:13:15.588475Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:15.588475Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Autoregressive Models in Vision: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chaofan Tao, Chengyue Wu, Gongye Liu, Guillermo Sapiro, Hongxia Yang, Huaxiu Yao, Hui Shen, Jiebo Luo, Jinfa Huang, Jing Xiong, Lingpeng Kong, Lun Huang, Mi Zhang, Ngai Wong, Ping Luo, Shen Yan, Taiqiang Wu, Yao Mu, Yuan Yao, Zhongwei Wan","submitted_at":"2024-11-08T17:15:12Z","abstract_excerpt":"Autoregressive modeling has been a huge success in the field of natural language processing (NLP). Recently, autoregressive models have emerged as a significant area of focus in computer vision, where they excel in producing high-quality visual content. Autoregressive models in NLP typically operate on subword tokens. However, the representation strategy in computer vision can vary in different levels, i.e., pixel-level, token-level, or scale-level, reflecting the diverse and hierarchical nature of visual data compared to the sequential structure of language. This survey comprehensively examin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.05902","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.05902/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.05902","created_at":"2026-07-05T11:13:15.588554+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.05902v2","created_at":"2026-07-05T11:13:15.588554+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.05902","created_at":"2026-07-05T11:13:15.588554+00:00"},{"alias_kind":"pith_short_12","alias_value":"DL7TW56R37CE","created_at":"2026-07-05T11:13:15.588554+00:00"},{"alias_kind":"pith_short_16","alias_value":"DL7TW56R37CEWLKS","created_at":"2026-07-05T11:13:15.588554+00:00"},{"alias_kind":"pith_short_8","alias_value":"DL7TW56R","created_at":"2026-07-05T11:13:15.588554+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27147","citing_title":"Safe Autoregressive Image Generation with Iterative Self-Improving Codebooks","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15015","citing_title":"NEXUS: Neural Energy Fields for Physically Consistent Contact-Rich 3D Object Dynamics","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01543","citing_title":"PathAR: Structure-First Autoregressive Synthesis of Multimodal Pathology Images","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2506.18601","citing_title":"BulletGen: Improving 4D Reconstruction with Bullet-Time Generation","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2511.08416","citing_title":"Generative AI Meets 6G and Beyond: Diffusion Models for Semantic Communications","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00814","citing_title":"Persistent Visual Memory: Sustaining Perception for Deep Generation in LVLMs","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00814","citing_title":"Persistent Visual Memory: Sustaining Perception for Deep Generation in LVLMs","ref_index":85,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DL7TW56R37CEWLKSL4MLDRDJLW","json":"https://pith.science/pith/DL7TW56R37CEWLKSL4MLDRDJLW.json","graph_json":"https://pith.science/api/pith-number/DL7TW56R37CEWLKSL4MLDRDJLW/graph.json","events_json":"https://pith.science/api/pith-number/DL7TW56R37CEWLKSL4MLDRDJLW/events.json","paper":"https://pith.science/paper/DL7TW56R"},"agent_actions":{"view_html":"https://pith.science/pith/DL7TW56R37CEWLKSL4MLDRDJLW","download_json":"https://pith.science/pith/DL7TW56R37CEWLKSL4MLDRDJLW.json","view_paper":"https://pith.science/paper/DL7TW56R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.05902&json=true","fetch_graph":"https://pith.science/api/pith-number/DL7TW56R37CEWLKSL4MLDRDJLW/graph.json","fetch_events":"https://pith.science/api/pith-number/DL7TW56R37CEWLKSL4MLDRDJLW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DL7TW56R37CEWLKSL4MLDRDJLW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DL7TW56R37CEWLKSL4MLDRDJLW/action/storage_attestation","attest_author":"https://pith.science/pith/DL7TW56R37CEWLKSL4MLDRDJLW/action/author_attestation","sign_citation":"https://pith.science/pith/DL7TW56R37CEWLKSL4MLDRDJLW/action/citation_signature","submit_replication":"https://pith.science/pith/DL7TW56R37CEWLKSL4MLDRDJLW/action/replication_record"}},"created_at":"2026-07-05T11:13:15.588554+00:00","updated_at":"2026-07-05T11:13:15.588554+00:00"}