{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VAHSMXBKKFL3HTXIKDX2B4N6YU","short_pith_number":"pith:VAHSMXBK","schema_version":"1.0","canonical_sha256":"a80f265c2a5157b3cee850efa0f1bec50f6f4dd066435fbc8d4846dd77d0cf2e","source":{"kind":"arxiv","id":"2505.10562","version":1},"attestation_state":"computed","paper":{"title":"End-to-End Vision Tokenizer Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fan Zhang, Haiwen Diao, Huchuan Lu, Jing Liu, Wenxuan Wang, Xinlong Wang, Yufeng Cui, Zhuoyan Luo","submitted_at":"2025-05-15T17:59:39Z","abstract_excerpt":"Existing vision tokenization isolates the optimization of vision tokenizers from downstream training, implicitly assuming the visual tokens can generalize well across various tasks, e.g., image generation and visual question answering. The vision tokenizer optimized for low-level reconstruction is agnostic to downstream tasks requiring varied representations and semantics. This decoupled paradigm introduces a critical misalignment: The loss of the vision tokenization can be the representation bottleneck for target tasks. For example, errors in tokenizing text in a given image lead to poor resu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.10562","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-15T17:59:39Z","cross_cats_sorted":[],"title_canon_sha256":"817017249d0e86997b03dad12e81a61f4fbc047cbc965b23f62a7f2fe9d49a34","abstract_canon_sha256":"75d74e82f86712ca3eaae45d7bf2cf42c62b75f0f6476618da0fb48a4ff6c122"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:45.782386Z","signature_b64":"h7fuec4CH5DDHlorvOwnI/AT3XvkBHMJemlNPtyYc+RMFlKi/vGbpzyOhdR73jADWkGaaKxKbRROXoAlWXtCAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a80f265c2a5157b3cee850efa0f1bec50f6f4dd066435fbc8d4846dd77d0cf2e","last_reissued_at":"2026-07-05T11:03:45.781891Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:45.781891Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"End-to-End Vision Tokenizer Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fan Zhang, Haiwen Diao, Huchuan Lu, Jing Liu, Wenxuan Wang, Xinlong Wang, Yufeng Cui, Zhuoyan Luo","submitted_at":"2025-05-15T17:59:39Z","abstract_excerpt":"Existing vision tokenization isolates the optimization of vision tokenizers from downstream training, implicitly assuming the visual tokens can generalize well across various tasks, e.g., image generation and visual question answering. The vision tokenizer optimized for low-level reconstruction is agnostic to downstream tasks requiring varied representations and semantics. This decoupled paradigm introduces a critical misalignment: The loss of the vision tokenization can be the representation bottleneck for target tasks. For example, errors in tokenizing text in a given image lead to poor resu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.10562","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.10562/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.10562","created_at":"2026-07-05T11:03:45.781949+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.10562v1","created_at":"2026-07-05T11:03:45.781949+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.10562","created_at":"2026-07-05T11:03:45.781949+00:00"},{"alias_kind":"pith_short_12","alias_value":"VAHSMXBKKFL3","created_at":"2026-07-05T11:03:45.781949+00:00"},{"alias_kind":"pith_short_16","alias_value":"VAHSMXBKKFL3HTXI","created_at":"2026-07-05T11:03:45.781949+00:00"},{"alias_kind":"pith_short_8","alias_value":"VAHSMXBK","created_at":"2026-07-05T11:03:45.781949+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03578","citing_title":"Diffusing in the Right Space: A Systematic Study of Latent Diffusability","ref_index":112,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VAHSMXBKKFL3HTXIKDX2B4N6YU","json":"https://pith.science/pith/VAHSMXBKKFL3HTXIKDX2B4N6YU.json","graph_json":"https://pith.science/api/pith-number/VAHSMXBKKFL3HTXIKDX2B4N6YU/graph.json","events_json":"https://pith.science/api/pith-number/VAHSMXBKKFL3HTXIKDX2B4N6YU/events.json","paper":"https://pith.science/paper/VAHSMXBK"},"agent_actions":{"view_html":"https://pith.science/pith/VAHSMXBKKFL3HTXIKDX2B4N6YU","download_json":"https://pith.science/pith/VAHSMXBKKFL3HTXIKDX2B4N6YU.json","view_paper":"https://pith.science/paper/VAHSMXBK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.10562&json=true","fetch_graph":"https://pith.science/api/pith-number/VAHSMXBKKFL3HTXIKDX2B4N6YU/graph.json","fetch_events":"https://pith.science/api/pith-number/VAHSMXBKKFL3HTXIKDX2B4N6YU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VAHSMXBKKFL3HTXIKDX2B4N6YU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VAHSMXBKKFL3HTXIKDX2B4N6YU/action/storage_attestation","attest_author":"https://pith.science/pith/VAHSMXBKKFL3HTXIKDX2B4N6YU/action/author_attestation","sign_citation":"https://pith.science/pith/VAHSMXBKKFL3HTXIKDX2B4N6YU/action/citation_signature","submit_replication":"https://pith.science/pith/VAHSMXBKKFL3HTXIKDX2B4N6YU/action/replication_record"}},"created_at":"2026-07-05T11:03:45.781949+00:00","updated_at":"2026-07-05T11:03:45.781949+00:00"}