{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OF5CSHHU5EIMJ5SYAXS6K3OZ3C","short_pith_number":"pith:OF5CSHHU","schema_version":"1.0","canonical_sha256":"717a291cf4e910c4f65805e5e56dd9d8b8c99de9ee6c8b80bd3a410f4f301fa9","source":{"kind":"arxiv","id":"2507.02358","version":4},"attestation_state":"computed","paper":{"title":"Hita: Holistic Tokenizer for Autoregressive Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Anlin Zheng, Haochen Wang, Tiancai Wang, Weipeng Deng, Xiangyu Zhang, Xiaojuan Qi, Yucheng Zhao","submitted_at":"2025-07-03T06:44:26Z","abstract_excerpt":"Vanilla autoregressive image generation models generate visual tokens step-by-step, limiting their ability to capture holistic relationships among token sequences. Moreover, because most visual tokenizers map local image patches into latent tokens, global information is limited. To address this, we introduce \\textit{Hita}, a novel image tokenizer for autoregressive (AR) image generation. It introduces a holistic-to-local tokenization scheme with learnable holistic queries and local patch tokens. Hita incorporates two key strategies to better align with the AR generation process: 1) {arranging}"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.02358","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-03T06:44:26Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0e55d309edf92f60c086ae141b1a5dd6a229c7a768396846415bbd66cb33f9db","abstract_canon_sha256":"05729221029683fbc524153eb27bccb905de2f24a82bb53a55c5f97efd23914b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:35:36.071054Z","signature_b64":"cnJSs3oDXV0fhoI+NMKui4Vrd0GNwCphvxx/ANBZ86IIyxO3XPbJcKNphvFH/TCxihfx7anYk87mA6YbHT9DCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"717a291cf4e910c4f65805e5e56dd9d8b8c99de9ee6c8b80bd3a410f4f301fa9","last_reissued_at":"2026-07-05T11:35:36.070651Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:35:36.070651Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hita: Holistic Tokenizer for Autoregressive Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Anlin Zheng, Haochen Wang, Tiancai Wang, Weipeng Deng, Xiangyu Zhang, Xiaojuan Qi, Yucheng Zhao","submitted_at":"2025-07-03T06:44:26Z","abstract_excerpt":"Vanilla autoregressive image generation models generate visual tokens step-by-step, limiting their ability to capture holistic relationships among token sequences. Moreover, because most visual tokenizers map local image patches into latent tokens, global information is limited. To address this, we introduce \\textit{Hita}, a novel image tokenizer for autoregressive (AR) image generation. It introduces a holistic-to-local tokenization scheme with learnable holistic queries and local patch tokens. Hita incorporates two key strategies to better align with the AR generation process: 1) {arranging}"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.02358","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.02358/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.02358","created_at":"2026-07-05T11:35:36.070709+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.02358v4","created_at":"2026-07-05T11:35:36.070709+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.02358","created_at":"2026-07-05T11:35:36.070709+00:00"},{"alias_kind":"pith_short_12","alias_value":"OF5CSHHU5EIM","created_at":"2026-07-05T11:35:36.070709+00:00"},{"alias_kind":"pith_short_16","alias_value":"OF5CSHHU5EIMJ5SY","created_at":"2026-07-05T11:35:36.070709+00:00"},{"alias_kind":"pith_short_8","alias_value":"OF5CSHHU","created_at":"2026-07-05T11:35:36.070709+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06137","citing_title":"Autoregressive Visual Generation Needs a Prologue","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2601.01593","citing_title":"Beyond Patches: Global-aware Autoregressive Model for Multimodal Few-Shot Font Generation","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16384","citing_title":"Mutual Enhancement Between Global Tokens and Patch Tokens: From Theory to Practice","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06137","citing_title":"Autoregressive Visual Generation Needs a Prologue","ref_index":64,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OF5CSHHU5EIMJ5SYAXS6K3OZ3C","json":"https://pith.science/pith/OF5CSHHU5EIMJ5SYAXS6K3OZ3C.json","graph_json":"https://pith.science/api/pith-number/OF5CSHHU5EIMJ5SYAXS6K3OZ3C/graph.json","events_json":"https://pith.science/api/pith-number/OF5CSHHU5EIMJ5SYAXS6K3OZ3C/events.json","paper":"https://pith.science/paper/OF5CSHHU"},"agent_actions":{"view_html":"https://pith.science/pith/OF5CSHHU5EIMJ5SYAXS6K3OZ3C","download_json":"https://pith.science/pith/OF5CSHHU5EIMJ5SYAXS6K3OZ3C.json","view_paper":"https://pith.science/paper/OF5CSHHU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.02358&json=true","fetch_graph":"https://pith.science/api/pith-number/OF5CSHHU5EIMJ5SYAXS6K3OZ3C/graph.json","fetch_events":"https://pith.science/api/pith-number/OF5CSHHU5EIMJ5SYAXS6K3OZ3C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OF5CSHHU5EIMJ5SYAXS6K3OZ3C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OF5CSHHU5EIMJ5SYAXS6K3OZ3C/action/storage_attestation","attest_author":"https://pith.science/pith/OF5CSHHU5EIMJ5SYAXS6K3OZ3C/action/author_attestation","sign_citation":"https://pith.science/pith/OF5CSHHU5EIMJ5SYAXS6K3OZ3C/action/citation_signature","submit_replication":"https://pith.science/pith/OF5CSHHU5EIMJ5SYAXS6K3OZ3C/action/replication_record"}},"created_at":"2026-07-05T11:35:36.070709+00:00","updated_at":"2026-07-05T11:35:36.070709+00:00"}