{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TYGM2W6NQGGKCZ2I7R2QFBEMJW","short_pith_number":"pith:TYGM2W6N","schema_version":"1.0","canonical_sha256":"9e0ccd5bcd818ca16748fc7502848c4d8dc6023f6e9e0aabf35b7ba5b5da6ccc","source":{"kind":"arxiv","id":"2505.13439","version":1},"attestation_state":"computed","paper":{"title":"VTBench: Evaluating Visual Tokenizers for Autoregressive Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Huawei Lin, Tong Geng, Weijie Zhao, Zhaozhuo Xu","submitted_at":"2025-05-19T17:59:01Z","abstract_excerpt":"Autoregressive (AR) models have recently shown strong performance in image generation, where a critical component is the visual tokenizer (VT) that maps continuous pixel inputs to discrete token sequences. The quality of the VT largely defines the upper bound of AR model performance. However, current discrete VTs fall significantly behind continuous variational autoencoders (VAEs), leading to degraded image reconstructions and poor preservation of details and text. Existing benchmarks focus on end-to-end generation quality, without isolating VT performance. To address this gap, we introduce VT"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.13439","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-19T17:59:01Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b225256d88b7142370cf0f0ab3ab8e7ce36b11783339e2c0f80187e21052c0d4","abstract_canon_sha256":"6f859471f6fa505adbfdf05cb101870782fbcb889c033c4467c15a9d7e39ee6d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:30.187332Z","signature_b64":"eme+IeE2DDk2kBOS5L9LKuomvbiu6ceVdXyfGjVZuDgQIoflbT3IN6l91XJ6wyN4MHdppG2nfM6mAG8osWoCBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e0ccd5bcd818ca16748fc7502848c4d8dc6023f6e9e0aabf35b7ba5b5da6ccc","last_reissued_at":"2026-07-05T11:05:30.186835Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:30.186835Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VTBench: Evaluating Visual Tokenizers for Autoregressive Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Huawei Lin, Tong Geng, Weijie Zhao, Zhaozhuo Xu","submitted_at":"2025-05-19T17:59:01Z","abstract_excerpt":"Autoregressive (AR) models have recently shown strong performance in image generation, where a critical component is the visual tokenizer (VT) that maps continuous pixel inputs to discrete token sequences. The quality of the VT largely defines the upper bound of AR model performance. However, current discrete VTs fall significantly behind continuous variational autoencoders (VAEs), leading to degraded image reconstructions and poor preservation of details and text. Existing benchmarks focus on end-to-end generation quality, without isolating VT performance. To address this gap, we introduce VT"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.13439","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.13439/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.13439","created_at":"2026-07-05T11:05:30.186893+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.13439v1","created_at":"2026-07-05T11:05:30.186893+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.13439","created_at":"2026-07-05T11:05:30.186893+00:00"},{"alias_kind":"pith_short_12","alias_value":"TYGM2W6NQGGK","created_at":"2026-07-05T11:05:30.186893+00:00"},{"alias_kind":"pith_short_16","alias_value":"TYGM2W6NQGGKCZ2I","created_at":"2026-07-05T11:05:30.186893+00:00"},{"alias_kind":"pith_short_8","alias_value":"TYGM2W6N","created_at":"2026-07-05T11:05:30.186893+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14333","citing_title":"InsightTok: Improving Text and Face Fidelity in Discrete Tokenization for Autoregressive Image Generation","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TYGM2W6NQGGKCZ2I7R2QFBEMJW","json":"https://pith.science/pith/TYGM2W6NQGGKCZ2I7R2QFBEMJW.json","graph_json":"https://pith.science/api/pith-number/TYGM2W6NQGGKCZ2I7R2QFBEMJW/graph.json","events_json":"https://pith.science/api/pith-number/TYGM2W6NQGGKCZ2I7R2QFBEMJW/events.json","paper":"https://pith.science/paper/TYGM2W6N"},"agent_actions":{"view_html":"https://pith.science/pith/TYGM2W6NQGGKCZ2I7R2QFBEMJW","download_json":"https://pith.science/pith/TYGM2W6NQGGKCZ2I7R2QFBEMJW.json","view_paper":"https://pith.science/paper/TYGM2W6N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.13439&json=true","fetch_graph":"https://pith.science/api/pith-number/TYGM2W6NQGGKCZ2I7R2QFBEMJW/graph.json","fetch_events":"https://pith.science/api/pith-number/TYGM2W6NQGGKCZ2I7R2QFBEMJW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TYGM2W6NQGGKCZ2I7R2QFBEMJW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TYGM2W6NQGGKCZ2I7R2QFBEMJW/action/storage_attestation","attest_author":"https://pith.science/pith/TYGM2W6NQGGKCZ2I7R2QFBEMJW/action/author_attestation","sign_citation":"https://pith.science/pith/TYGM2W6NQGGKCZ2I7R2QFBEMJW/action/citation_signature","submit_replication":"https://pith.science/pith/TYGM2W6NQGGKCZ2I7R2QFBEMJW/action/replication_record"}},"created_at":"2026-07-05T11:05:30.186893+00:00","updated_at":"2026-07-05T11:05:30.186893+00:00"}