{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KOK7SOFPKBBA7ZSJMY6BHM6BUB","short_pith_number":"pith:KOK7SOFP","schema_version":"1.0","canonical_sha256":"5395f938af50420fe649663c13b3c1a05a7c9258e8a508de452db3971f78edc3","source":{"kind":"arxiv","id":"2506.10962","version":1},"attestation_state":"computed","paper":{"title":"SpectralAR: Spectral Autoregressive Visual Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Jie Zhou, Jiwen Lu, Weiliang Chen, Wenzhao Zheng, Yuanhui Huang, Yueqi Duan","submitted_at":"2025-06-12T17:57:44Z","abstract_excerpt":"Autoregressive visual generation has garnered increasing attention due to its scalability and compatibility with other modalities compared with diffusion models. Most existing methods construct visual sequences as spatial patches for autoregressive generation. However, image patches are inherently parallel, contradicting the causal nature of autoregressive modeling. To address this, we propose a Spectral AutoRegressive (SpectralAR) visual generation framework, which realizes causality for visual sequences from the spectral perspective. Specifically, we first transform an image into ordered spe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.10962","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-12T17:57:44Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"77aa57189f0407e50193751ea9e6e76aedfc8aedf4a62371927d62ee71f5e45c","abstract_canon_sha256":"4ab834f6244ac9501bbf5709978a2ab8f4d2f259fc1f0ba39993633d0fa2e3a9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:41.171211Z","signature_b64":"6x+R/XyIhYKXlv5azX/XS/keU9hirScm/T88GwG7oX2+hNMn7W/IrQ8pE11jxpWVbvYUhwld1e/D4c998MRdAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5395f938af50420fe649663c13b3c1a05a7c9258e8a508de452db3971f78edc3","last_reissued_at":"2026-07-05T11:20:41.170761Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:41.170761Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SpectralAR: Spectral Autoregressive Visual Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Jie Zhou, Jiwen Lu, Weiliang Chen, Wenzhao Zheng, Yuanhui Huang, Yueqi Duan","submitted_at":"2025-06-12T17:57:44Z","abstract_excerpt":"Autoregressive visual generation has garnered increasing attention due to its scalability and compatibility with other modalities compared with diffusion models. Most existing methods construct visual sequences as spatial patches for autoregressive generation. However, image patches are inherently parallel, contradicting the causal nature of autoregressive modeling. To address this, we propose a Spectral AutoRegressive (SpectralAR) visual generation framework, which realizes causality for visual sequences from the spectral perspective. Specifically, we first transform an image into ordered spe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.10962","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.10962/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.10962","created_at":"2026-07-05T11:20:41.170823+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.10962v1","created_at":"2026-07-05T11:20:41.170823+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.10962","created_at":"2026-07-05T11:20:41.170823+00:00"},{"alias_kind":"pith_short_12","alias_value":"KOK7SOFPKBBA","created_at":"2026-07-05T11:20:41.170823+00:00"},{"alias_kind":"pith_short_16","alias_value":"KOK7SOFPKBBA7ZSJ","created_at":"2026-07-05T11:20:41.170823+00:00"},{"alias_kind":"pith_short_8","alias_value":"KOK7SOFP","created_at":"2026-07-05T11:20:41.170823+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17590","citing_title":"TivTok: Broadcasting Time-Invariant Tokens for Scalable Video Tokenization","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00371","citing_title":"MEPA: Multi-Scale Representation Alignment for Visual Autoregressive Modeling with Mixture of Experts","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06137","citing_title":"Autoregressive Visual Generation Needs a Prologue","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2601.01593","citing_title":"Beyond Patches: Global-aware Autoregressive Model for Multimodal Few-Shot Font Generation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06137","citing_title":"Autoregressive Visual Generation Needs a Prologue","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00503","citing_title":"End-to-End Autoregressive Image Generation with 1D Semantic Tokenizer","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KOK7SOFPKBBA7ZSJMY6BHM6BUB","json":"https://pith.science/pith/KOK7SOFPKBBA7ZSJMY6BHM6BUB.json","graph_json":"https://pith.science/api/pith-number/KOK7SOFPKBBA7ZSJMY6BHM6BUB/graph.json","events_json":"https://pith.science/api/pith-number/KOK7SOFPKBBA7ZSJMY6BHM6BUB/events.json","paper":"https://pith.science/paper/KOK7SOFP"},"agent_actions":{"view_html":"https://pith.science/pith/KOK7SOFPKBBA7ZSJMY6BHM6BUB","download_json":"https://pith.science/pith/KOK7SOFPKBBA7ZSJMY6BHM6BUB.json","view_paper":"https://pith.science/paper/KOK7SOFP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.10962&json=true","fetch_graph":"https://pith.science/api/pith-number/KOK7SOFPKBBA7ZSJMY6BHM6BUB/graph.json","fetch_events":"https://pith.science/api/pith-number/KOK7SOFPKBBA7ZSJMY6BHM6BUB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KOK7SOFPKBBA7ZSJMY6BHM6BUB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KOK7SOFPKBBA7ZSJMY6BHM6BUB/action/storage_attestation","attest_author":"https://pith.science/pith/KOK7SOFPKBBA7ZSJMY6BHM6BUB/action/author_attestation","sign_citation":"https://pith.science/pith/KOK7SOFPKBBA7ZSJMY6BHM6BUB/action/citation_signature","submit_replication":"https://pith.science/pith/KOK7SOFPKBBA7ZSJMY6BHM6BUB/action/replication_record"}},"created_at":"2026-07-05T11:20:41.170823+00:00","updated_at":"2026-07-05T11:20:41.170823+00:00"}