{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YT2NLIULMEVGDIT7TM7XZGFA4Y","short_pith_number":"pith:YT2NLIUL","schema_version":"1.0","canonical_sha256":"c4f4d5a28b612a61a27f9b3f7c98a0e621bc8b9749dfeb5325c0d20522d9b2be","source":{"kind":"arxiv","id":"2502.05178","version":1},"attestation_state":"computed","paper":{"title":"QLIP: Text-Aligned Visual Tokenization Unifies Auto-Regressive Multimodal Understanding and Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"De-An Huang, Fuzhao Xue, Jan Kautz, Linxi Fan, Philipp Kr\\\"ahenb\\\"uhl, Scott Reed, Yue Zhao, Yuke Zhu, Zhiding Yu","submitted_at":"2025-02-07T18:59:57Z","abstract_excerpt":"We introduce Quantized Language-Image Pretraining (QLIP), a visual tokenization method that combines state-of-the-art reconstruction quality with state-of-the-art zero-shot image understanding. QLIP trains a binary-spherical-quantization-based autoencoder with reconstruction and language-image alignment objectives. We are the first to show that the two objectives do not need to be at odds. We balance the two loss terms dynamically during training and show that a two-stage training pipeline effectively mixes the large-batch requirements of image-language pre-training with the memory bottleneck "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.05178","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-07T18:59:57Z","cross_cats_sorted":[],"title_canon_sha256":"78132c63ef322cfec46d5dde3007b01478345ece6f41395a8c206adb1eac49f8","abstract_canon_sha256":"9e9e1a4a7ff041658e50d7731d1b85c0834232fa45b102005acadafad67e9977"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:10.343396Z","signature_b64":"j0lP/bGBpPx0f96t0lNnMzAVZoStk0urlF+elI6afzL7X/ILXukw8u6NNI101A9uk9lLoREbYPMI8YWWFqUJCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c4f4d5a28b612a61a27f9b3f7c98a0e621bc8b9749dfeb5325c0d20522d9b2be","last_reissued_at":"2026-07-05T10:11:10.342890Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:10.342890Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"QLIP: Text-Aligned Visual Tokenization Unifies Auto-Regressive Multimodal Understanding and Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"De-An Huang, Fuzhao Xue, Jan Kautz, Linxi Fan, Philipp Kr\\\"ahenb\\\"uhl, Scott Reed, Yue Zhao, Yuke Zhu, Zhiding Yu","submitted_at":"2025-02-07T18:59:57Z","abstract_excerpt":"We introduce Quantized Language-Image Pretraining (QLIP), a visual tokenization method that combines state-of-the-art reconstruction quality with state-of-the-art zero-shot image understanding. QLIP trains a binary-spherical-quantization-based autoencoder with reconstruction and language-image alignment objectives. We are the first to show that the two objectives do not need to be at odds. We balance the two loss terms dynamically during training and show that a two-stage training pipeline effectively mixes the large-batch requirements of image-language pre-training with the memory bottleneck "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.05178","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.05178/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.05178","created_at":"2026-07-05T10:11:10.342954+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.05178v1","created_at":"2026-07-05T10:11:10.342954+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.05178","created_at":"2026-07-05T10:11:10.342954+00:00"},{"alias_kind":"pith_short_12","alias_value":"YT2NLIULMEVG","created_at":"2026-07-05T10:11:10.342954+00:00"},{"alias_kind":"pith_short_16","alias_value":"YT2NLIULMEVGDIT7","created_at":"2026-07-05T10:11:10.342954+00:00"},{"alias_kind":"pith_short_8","alias_value":"YT2NLIUL","created_at":"2026-07-05T10:11:10.342954+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18249","citing_title":"Unified Multimodal Autoregressive Modeling with Shared Context-Visual Tokenizer is Key to Unification","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13289","citing_title":"HYDRA-X: Native Unified Multimodal Models with Holistic Visual Tokenizers","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11363","citing_title":"NSVQ: Mitigating Codebook Collapse by Stabilizing Encoder Drift in Vector Quantization","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03578","citing_title":"Diffusing in the Right Space: A Systematic Study of Latent Diffusability","ref_index":113,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14028","citing_title":"Unified Pix Token And Word Token Generative Language Model","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18115","citing_title":"WinTok: A Win-Win Hybrid Tokenizer via Decomposing Visual Understanding and Generation with Transferable Tokens","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12500","citing_title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","ref_index":167,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YT2NLIULMEVGDIT7TM7XZGFA4Y","json":"https://pith.science/pith/YT2NLIULMEVGDIT7TM7XZGFA4Y.json","graph_json":"https://pith.science/api/pith-number/YT2NLIULMEVGDIT7TM7XZGFA4Y/graph.json","events_json":"https://pith.science/api/pith-number/YT2NLIULMEVGDIT7TM7XZGFA4Y/events.json","paper":"https://pith.science/paper/YT2NLIUL"},"agent_actions":{"view_html":"https://pith.science/pith/YT2NLIULMEVGDIT7TM7XZGFA4Y","download_json":"https://pith.science/pith/YT2NLIULMEVGDIT7TM7XZGFA4Y.json","view_paper":"https://pith.science/paper/YT2NLIUL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.05178&json=true","fetch_graph":"https://pith.science/api/pith-number/YT2NLIULMEVGDIT7TM7XZGFA4Y/graph.json","fetch_events":"https://pith.science/api/pith-number/YT2NLIULMEVGDIT7TM7XZGFA4Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YT2NLIULMEVGDIT7TM7XZGFA4Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YT2NLIULMEVGDIT7TM7XZGFA4Y/action/storage_attestation","attest_author":"https://pith.science/pith/YT2NLIULMEVGDIT7TM7XZGFA4Y/action/author_attestation","sign_citation":"https://pith.science/pith/YT2NLIULMEVGDIT7TM7XZGFA4Y/action/citation_signature","submit_replication":"https://pith.science/pith/YT2NLIULMEVGDIT7TM7XZGFA4Y/action/replication_record"}},"created_at":"2026-07-05T10:11:10.342954+00:00","updated_at":"2026-07-05T10:11:10.342954+00:00"}