{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:D4ZPIPZOQNQQQLGUP74NGLRBJM","short_pith_number":"pith:D4ZPIPZO","schema_version":"1.0","canonical_sha256":"1f32f43f2e8361082cd47ff8d32e214b0378fb6d96bca20c20f93285c793a7a3","source":{"kind":"arxiv","id":"2504.08736","version":2},"attestation_state":"computed","paper":{"title":"GigaTok: Scaling Visual Tokenizers to 3 Billion Parameters for Autoregressive Image Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiashi Feng, Jun Hao Liew, Tianwei Xiong, Xihui Liu, Zilong Huang","submitted_at":"2025-04-11T17:59:58Z","abstract_excerpt":"In autoregressive (AR) image generation, visual tokenizers compress images into compact discrete latent tokens, enabling efficient training of downstream autoregressive models for visual generation via next-token prediction. While scaling visual tokenizers improves image reconstruction quality, it often degrades downstream generation quality -- a challenge not adequately addressed in existing literature. To address this, we introduce GigaTok, the first approach to simultaneously improve image reconstruction, generation, and representation learning when scaling visual tokenizers. We identify th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.08736","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-11T17:59:58Z","cross_cats_sorted":[],"title_canon_sha256":"cc28d0d711347cbaba42a08a279afe053000b310808fbae10f63433cb487e0af","abstract_canon_sha256":"ca0b4303df25c3ecf4f67726b07732c934e2119081da357b156208178e879887"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:58:27.355451Z","signature_b64":"QyLv/Jgq+BRDWyN0xfYlqtyF2eaMbQ/p9b6g3c2BF/3D+vkfqXgeVCg+zJkbGkipqCfMhkfY/2Uc1aO85IDCDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f32f43f2e8361082cd47ff8d32e214b0378fb6d96bca20c20f93285c793a7a3","last_reissued_at":"2026-07-05T11:58:27.354966Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:58:27.354966Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GigaTok: Scaling Visual Tokenizers to 3 Billion Parameters for Autoregressive Image Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiashi Feng, Jun Hao Liew, Tianwei Xiong, Xihui Liu, Zilong Huang","submitted_at":"2025-04-11T17:59:58Z","abstract_excerpt":"In autoregressive (AR) image generation, visual tokenizers compress images into compact discrete latent tokens, enabling efficient training of downstream autoregressive models for visual generation via next-token prediction. While scaling visual tokenizers improves image reconstruction quality, it often degrades downstream generation quality -- a challenge not adequately addressed in existing literature. To address this, we introduce GigaTok, the first approach to simultaneously improve image reconstruction, generation, and representation learning when scaling visual tokenizers. We identify th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.08736","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.08736/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.08736","created_at":"2026-07-05T11:58:27.355023+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.08736v2","created_at":"2026-07-05T11:58:27.355023+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.08736","created_at":"2026-07-05T11:58:27.355023+00:00"},{"alias_kind":"pith_short_12","alias_value":"D4ZPIPZOQNQQ","created_at":"2026-07-05T11:58:27.355023+00:00"},{"alias_kind":"pith_short_16","alias_value":"D4ZPIPZOQNQQQLGU","created_at":"2026-07-05T11:58:27.355023+00:00"},{"alias_kind":"pith_short_8","alias_value":"D4ZPIPZO","created_at":"2026-07-05T11:58:27.355023+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05552","citing_title":"Balancing Image Compression and Generation with Bootstrapped Tokenization","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06137","citing_title":"Autoregressive Visual Generation Needs a Prologue","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06137","citing_title":"Autoregressive Visual Generation Needs a Prologue","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00503","citing_title":"End-to-End Autoregressive Image Generation with 1D Semantic Tokenizer","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07340","citing_title":"TC-AE: Unlocking Token Capacity for Deep Compression Autoencoders","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D4ZPIPZOQNQQQLGUP74NGLRBJM","json":"https://pith.science/pith/D4ZPIPZOQNQQQLGUP74NGLRBJM.json","graph_json":"https://pith.science/api/pith-number/D4ZPIPZOQNQQQLGUP74NGLRBJM/graph.json","events_json":"https://pith.science/api/pith-number/D4ZPIPZOQNQQQLGUP74NGLRBJM/events.json","paper":"https://pith.science/paper/D4ZPIPZO"},"agent_actions":{"view_html":"https://pith.science/pith/D4ZPIPZOQNQQQLGUP74NGLRBJM","download_json":"https://pith.science/pith/D4ZPIPZOQNQQQLGUP74NGLRBJM.json","view_paper":"https://pith.science/paper/D4ZPIPZO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.08736&json=true","fetch_graph":"https://pith.science/api/pith-number/D4ZPIPZOQNQQQLGUP74NGLRBJM/graph.json","fetch_events":"https://pith.science/api/pith-number/D4ZPIPZOQNQQQLGUP74NGLRBJM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D4ZPIPZOQNQQQLGUP74NGLRBJM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D4ZPIPZOQNQQQLGUP74NGLRBJM/action/storage_attestation","attest_author":"https://pith.science/pith/D4ZPIPZOQNQQQLGUP74NGLRBJM/action/author_attestation","sign_citation":"https://pith.science/pith/D4ZPIPZOQNQQQLGUP74NGLRBJM/action/citation_signature","submit_replication":"https://pith.science/pith/D4ZPIPZOQNQQQLGUP74NGLRBJM/action/replication_record"}},"created_at":"2026-07-05T11:58:27.355023+00:00","updated_at":"2026-07-05T11:58:27.355023+00:00"}