{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CBKWDS42I3WG6XFANIT73MA64W","short_pith_number":"pith:CBKWDS42","schema_version":"1.0","canonical_sha256":"105561cb9a46ec6f5ca06a27fdb01ee5943924c3729666b12f0b6bac00d47068","source":{"kind":"arxiv","id":"2503.08377","version":3},"attestation_state":"computed","paper":{"title":"Layton: Latent Consistency Tokenizer for 1024-pixel Image Reconstruction and Generation by 256 Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haonan Lu, Qingsong Xie, Yanhao Zhang, Zhao Zhang, Zhe Huang, Zhenyu Yang","submitted_at":"2025-03-11T12:38:12Z","abstract_excerpt":"Image tokenization has significantly advanced visual generation and multimodal modeling, particularly when paired with autoregressive models. However, current methods face challenges in balancing efficiency and fidelity: high-resolution image reconstruction either requires an excessive number of tokens or compromises critical details through token reduction. To resolve this, we propose Latent Consistency Tokenizer (Layton) that bridges discrete visual tokens with the compact latent space of pre-trained Latent Diffusion Models (LDMs), enabling efficient representation of 1024x1024 images using "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.08377","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-11T12:38:12Z","cross_cats_sorted":[],"title_canon_sha256":"d98f85c3cf3774031e54102223eb4fde3d7172bb5e2674897204ed1492db8f7d","abstract_canon_sha256":"4a7f42030cd8dec7b59006dbec4cbe655aaac1bb3732ae7f8591b2ef764e5e22"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:31:02.322474Z","signature_b64":"UD/99fQyySckjAke0CrGCz6Qa+I2O+GfgjwcUkTnO834RTlXPkgJ8vmzPDeG7k+2DJ2El46FSAcGYAsszh52BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"105561cb9a46ec6f5ca06a27fdb01ee5943924c3729666b12f0b6bac00d47068","last_reissued_at":"2026-07-05T10:31:02.321962Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:31:02.321962Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Layton: Latent Consistency Tokenizer for 1024-pixel Image Reconstruction and Generation by 256 Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haonan Lu, Qingsong Xie, Yanhao Zhang, Zhao Zhang, Zhe Huang, Zhenyu Yang","submitted_at":"2025-03-11T12:38:12Z","abstract_excerpt":"Image tokenization has significantly advanced visual generation and multimodal modeling, particularly when paired with autoregressive models. However, current methods face challenges in balancing efficiency and fidelity: high-resolution image reconstruction either requires an excessive number of tokens or compromises critical details through token reduction. To resolve this, we propose Latent Consistency Tokenizer (Layton) that bridges discrete visual tokens with the compact latent space of pre-trained Latent Diffusion Models (LDMs), enabling efficient representation of 1024x1024 images using "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.08377","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.08377/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.08377","created_at":"2026-07-05T10:31:02.322021+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.08377v3","created_at":"2026-07-05T10:31:02.322021+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.08377","created_at":"2026-07-05T10:31:02.322021+00:00"},{"alias_kind":"pith_short_12","alias_value":"CBKWDS42I3WG","created_at":"2026-07-05T10:31:02.322021+00:00"},{"alias_kind":"pith_short_16","alias_value":"CBKWDS42I3WG6XFA","created_at":"2026-07-05T10:31:02.322021+00:00"},{"alias_kind":"pith_short_8","alias_value":"CBKWDS42","created_at":"2026-07-05T10:31:02.322021+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.24885","citing_title":"VibeToken: Scaling 1D Image Tokenizers and Autoregressive Models for Dynamic Resolution Generations","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20940","citing_title":"Sema: Semantic Transport for Real-Time Multimodal Agents","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CBKWDS42I3WG6XFANIT73MA64W","json":"https://pith.science/pith/CBKWDS42I3WG6XFANIT73MA64W.json","graph_json":"https://pith.science/api/pith-number/CBKWDS42I3WG6XFANIT73MA64W/graph.json","events_json":"https://pith.science/api/pith-number/CBKWDS42I3WG6XFANIT73MA64W/events.json","paper":"https://pith.science/paper/CBKWDS42"},"agent_actions":{"view_html":"https://pith.science/pith/CBKWDS42I3WG6XFANIT73MA64W","download_json":"https://pith.science/pith/CBKWDS42I3WG6XFANIT73MA64W.json","view_paper":"https://pith.science/paper/CBKWDS42","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.08377&json=true","fetch_graph":"https://pith.science/api/pith-number/CBKWDS42I3WG6XFANIT73MA64W/graph.json","fetch_events":"https://pith.science/api/pith-number/CBKWDS42I3WG6XFANIT73MA64W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CBKWDS42I3WG6XFANIT73MA64W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CBKWDS42I3WG6XFANIT73MA64W/action/storage_attestation","attest_author":"https://pith.science/pith/CBKWDS42I3WG6XFANIT73MA64W/action/author_attestation","sign_citation":"https://pith.science/pith/CBKWDS42I3WG6XFANIT73MA64W/action/citation_signature","submit_replication":"https://pith.science/pith/CBKWDS42I3WG6XFANIT73MA64W/action/replication_record"}},"created_at":"2026-07-05T10:31:02.322021+00:00","updated_at":"2026-07-05T10:31:02.322021+00:00"}