{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3NLVTSSA2LEDAFQVMUFPQQYWQM","short_pith_number":"pith:3NLVTSSA","schema_version":"1.0","canonical_sha256":"db5759ca40d2c8301615650af84316831305ba7d5d0afd8a3d2b5637b47b7ac7","source":{"kind":"arxiv","id":"2506.21022","version":1},"attestation_state":"computed","paper":{"title":"Instella-T2I: Pushing the Limits of 1D Discrete Latent Space Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Benran Hu, Emad Barsoum, Hao Chen, Jialian Wu, Jiang Liu, Xiaodong Yu, Ximeng Sun, Yusheng Su, Ze Wang, Zicheng Liu","submitted_at":"2025-06-26T05:48:36Z","abstract_excerpt":"Image tokenization plays a critical role in reducing the computational demands of modeling high-resolution images, significantly improving the efficiency of image and multimodal understanding and generation. Recent advances in 1D latent spaces have reduced the number of tokens required by eliminating the need for a 2D grid structure. In this paper, we further advance compact discrete image representation by introducing 1D binary image latents. By representing each image as a sequence of binary vectors, rather than using traditional one-hot codebook tokens, our approach preserves high-resolutio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.21022","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-26T05:48:36Z","cross_cats_sorted":[],"title_canon_sha256":"dfaea4e3c6706e2b060dcfc43b78171eb8a3a5d3560c8b06dab9bca6aa00f117","abstract_canon_sha256":"0f4876e03004cc3c9c3d7f60d214fd6844d8ca884da006f108007656b5e2e7a6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:38.213285Z","signature_b64":"suHdILqcPcIGXuRcxOsE7L73Wsosr6XJxtsYJKPT9M/o3rcU65ph5O9mmFBmHaaYLEb+hljjqXYnQ6m7vNXqCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"db5759ca40d2c8301615650af84316831305ba7d5d0afd8a3d2b5637b47b7ac7","last_reissued_at":"2026-07-05T11:27:38.212595Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:38.212595Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Instella-T2I: Pushing the Limits of 1D Discrete Latent Space Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Benran Hu, Emad Barsoum, Hao Chen, Jialian Wu, Jiang Liu, Xiaodong Yu, Ximeng Sun, Yusheng Su, Ze Wang, Zicheng Liu","submitted_at":"2025-06-26T05:48:36Z","abstract_excerpt":"Image tokenization plays a critical role in reducing the computational demands of modeling high-resolution images, significantly improving the efficiency of image and multimodal understanding and generation. Recent advances in 1D latent spaces have reduced the number of tokens required by eliminating the need for a 2D grid structure. In this paper, we further advance compact discrete image representation by introducing 1D binary image latents. By representing each image as a sequence of binary vectors, rather than using traditional one-hot codebook tokens, our approach preserves high-resolutio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21022","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21022/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.21022","created_at":"2026-07-05T11:27:38.212653+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.21022v1","created_at":"2026-07-05T11:27:38.212653+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21022","created_at":"2026-07-05T11:27:38.212653+00:00"},{"alias_kind":"pith_short_12","alias_value":"3NLVTSSA2LED","created_at":"2026-07-05T11:27:38.212653+00:00"},{"alias_kind":"pith_short_16","alias_value":"3NLVTSSA2LEDAFQV","created_at":"2026-07-05T11:27:38.212653+00:00"},{"alias_kind":"pith_short_8","alias_value":"3NLVTSSA","created_at":"2026-07-05T11:27:38.212653+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.01593","citing_title":"Beyond Patches: Global-aware Autoregressive Model for Multimodal Few-Shot Font Generation","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24885","citing_title":"VibeToken: Scaling 1D Image Tokenizers and Autoregressive Models for Dynamic Resolution Generations","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3NLVTSSA2LEDAFQVMUFPQQYWQM","json":"https://pith.science/pith/3NLVTSSA2LEDAFQVMUFPQQYWQM.json","graph_json":"https://pith.science/api/pith-number/3NLVTSSA2LEDAFQVMUFPQQYWQM/graph.json","events_json":"https://pith.science/api/pith-number/3NLVTSSA2LEDAFQVMUFPQQYWQM/events.json","paper":"https://pith.science/paper/3NLVTSSA"},"agent_actions":{"view_html":"https://pith.science/pith/3NLVTSSA2LEDAFQVMUFPQQYWQM","download_json":"https://pith.science/pith/3NLVTSSA2LEDAFQVMUFPQQYWQM.json","view_paper":"https://pith.science/paper/3NLVTSSA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.21022&json=true","fetch_graph":"https://pith.science/api/pith-number/3NLVTSSA2LEDAFQVMUFPQQYWQM/graph.json","fetch_events":"https://pith.science/api/pith-number/3NLVTSSA2LEDAFQVMUFPQQYWQM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3NLVTSSA2LEDAFQVMUFPQQYWQM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3NLVTSSA2LEDAFQVMUFPQQYWQM/action/storage_attestation","attest_author":"https://pith.science/pith/3NLVTSSA2LEDAFQVMUFPQQYWQM/action/author_attestation","sign_citation":"https://pith.science/pith/3NLVTSSA2LEDAFQVMUFPQQYWQM/action/citation_signature","submit_replication":"https://pith.science/pith/3NLVTSSA2LEDAFQVMUFPQQYWQM/action/replication_record"}},"created_at":"2026-07-05T11:27:38.212653+00:00","updated_at":"2026-07-05T11:27:38.212653+00:00"}