{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DDTFBREXHZSUFHI53RR6TEJ3ZM","short_pith_number":"pith:DDTFBREX","schema_version":"1.0","canonical_sha256":"18e650c4973e65429d1ddc63e9913bcb28090e5c951f7ed36511d5aa6dbfd7a6","source":{"kind":"arxiv","id":"2412.04431","version":2},"attestation_state":"computed","paper":{"title":"Infinity: Scaling Bitwise AutoRegressive Modeling for High-Resolution Image Synthesis","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyue Peng, Bin Yan, Jian Han, Jinlai Liu, Xiaobing Liu, Yi Jiang, Yuqi Zhang, Zehuan Yuan","submitted_at":"2024-12-05T18:53:02Z","abstract_excerpt":"We present Infinity, a Bitwise Visual AutoRegressive Modeling capable of generating high-resolution, photorealistic images following language instruction. Infinity redefines visual autoregressive model under a bitwise token prediction framework with an infinite-vocabulary tokenizer & classifier and bitwise self-correction mechanism, remarkably improving the generation capacity and details. By theoretically scaling the tokenizer vocabulary size to infinity and concurrently scaling the transformer size, our method significantly unleashes powerful scaling capabilities compared to vanilla VAR. Inf"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.04431","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-05T18:53:02Z","cross_cats_sorted":[],"title_canon_sha256":"971b349b2e1be1db7b841a4c5be3f3a18acddd5533c8c66934ae9b787331003a","abstract_canon_sha256":"d568ce839a53050cfdc7e4ae9b8b55f8cef07e593f2e96ba8c27f07b412e1586"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:34.547179Z","signature_b64":"LUszc8BUww+lsOvHLMCykleDUnLOBDLFDSrzLuVaKpalVSAtKbP6nvLxe/EzdaIHlA86845PU9TlnTXkTnaABA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"18e650c4973e65429d1ddc63e9913bcb28090e5c951f7ed36511d5aa6dbfd7a6","last_reissued_at":"2026-07-05T11:22:34.546693Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:34.546693Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Infinity: Scaling Bitwise AutoRegressive Modeling for High-Resolution Image Synthesis","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyue Peng, Bin Yan, Jian Han, Jinlai Liu, Xiaobing Liu, Yi Jiang, Yuqi Zhang, Zehuan Yuan","submitted_at":"2024-12-05T18:53:02Z","abstract_excerpt":"We present Infinity, a Bitwise Visual AutoRegressive Modeling capable of generating high-resolution, photorealistic images following language instruction. Infinity redefines visual autoregressive model under a bitwise token prediction framework with an infinite-vocabulary tokenizer & classifier and bitwise self-correction mechanism, remarkably improving the generation capacity and details. By theoretically scaling the tokenizer vocabulary size to infinity and concurrently scaling the transformer size, our method significantly unleashes powerful scaling capabilities compared to vanilla VAR. Inf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.04431","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.04431/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.04431","created_at":"2026-07-05T11:22:34.546752+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.04431v2","created_at":"2026-07-05T11:22:34.546752+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.04431","created_at":"2026-07-05T11:22:34.546752+00:00"},{"alias_kind":"pith_short_12","alias_value":"DDTFBREXHZSU","created_at":"2026-07-05T11:22:34.546752+00:00"},{"alias_kind":"pith_short_16","alias_value":"DDTFBREXHZSUFHI5","created_at":"2026-07-05T11:22:34.546752+00:00"},{"alias_kind":"pith_short_8","alias_value":"DDTFBREX","created_at":"2026-07-05T11:22:34.546752+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08302","citing_title":"HACK++: Towards More Effective Head-Aware Key-Value Compression for Efficient Visual Autoregressive Modeling","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2504.13378","citing_title":"SMPL-GPTexture: Dual-View 3D Human Texture Estimation using Text-to-Image Generation Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2505.05472","citing_title":"Mogao: An Omni Foundation Model for Interleaved Multi-Modal Generation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07265","citing_title":"WISE: A World Knowledge-Informed Semantic Evaluation for Text-to-Image Generation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12967","citing_title":"ImageAttributionBench: How Far Are We from Generalizable Attribution?","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20275","citing_title":"ImgEdit: A Unified Image Editing Dataset and Benchmark","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21450","citing_title":"VARestorer: One-Step VAR Distillation for Real-World Image Super-Resolution","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13030","citing_title":"Generative Refinement Networks for Visual Synthesis","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DDTFBREXHZSUFHI53RR6TEJ3ZM","json":"https://pith.science/pith/DDTFBREXHZSUFHI53RR6TEJ3ZM.json","graph_json":"https://pith.science/api/pith-number/DDTFBREXHZSUFHI53RR6TEJ3ZM/graph.json","events_json":"https://pith.science/api/pith-number/DDTFBREXHZSUFHI53RR6TEJ3ZM/events.json","paper":"https://pith.science/paper/DDTFBREX"},"agent_actions":{"view_html":"https://pith.science/pith/DDTFBREXHZSUFHI53RR6TEJ3ZM","download_json":"https://pith.science/pith/DDTFBREXHZSUFHI53RR6TEJ3ZM.json","view_paper":"https://pith.science/paper/DDTFBREX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.04431&json=true","fetch_graph":"https://pith.science/api/pith-number/DDTFBREXHZSUFHI53RR6TEJ3ZM/graph.json","fetch_events":"https://pith.science/api/pith-number/DDTFBREXHZSUFHI53RR6TEJ3ZM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DDTFBREXHZSUFHI53RR6TEJ3ZM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DDTFBREXHZSUFHI53RR6TEJ3ZM/action/storage_attestation","attest_author":"https://pith.science/pith/DDTFBREXHZSUFHI53RR6TEJ3ZM/action/author_attestation","sign_citation":"https://pith.science/pith/DDTFBREXHZSUFHI53RR6TEJ3ZM/action/citation_signature","submit_replication":"https://pith.science/pith/DDTFBREXHZSUFHI53RR6TEJ3ZM/action/replication_record"}},"created_at":"2026-07-05T11:22:34.546752+00:00","updated_at":"2026-07-05T11:22:34.546752+00:00"}