{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6ESL3M4CYYVNQ4PEBN4665JH2M","short_pith_number":"pith:6ESL3M4C","schema_version":"1.0","canonical_sha256":"f124bdb382c62ad871e40b79ef7527d3372c5c73a4df3203046638be0ebe2b06","source":{"kind":"arxiv","id":"2408.08459","version":2},"attestation_state":"computed","paper":{"title":"JPEG-LM: LLMs as Image Generators with Canonical Codec Representations","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Marjan Ghazvininejad, Pang Wei Koh, Xiaochuang Han, Yulia Tsvetkov","submitted_at":"2024-08-15T23:57:02Z","abstract_excerpt":"Recent work in image and video generation has been adopting the autoregressive LLM architecture due to its generality and potentially easy integration into multi-modal systems. The crux of applying autoregressive training in language generation to visual generation is discretization -- representing continuous data like images and videos as discrete tokens. Common methods of discretizing images and videos include modeling raw pixel values, which are prohibitively lengthy, or vector quantization, which requires convoluted pre-hoc training. In this work, we propose to directly model images and vi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.08459","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-15T23:57:02Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"e87e3d80836d8501dc6ba2992b79be63f7a8d6276775772c4ab574ef74ab3902","abstract_canon_sha256":"7beb63e36781187a6f513e32d87c61bb74d0d3c0313bbfbd8451e0852436c3cc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:57:24.023091Z","signature_b64":"jUrx4gLnNAgtWKUUUh9V2t32Fbb1Wn9yQUuZPJi/aWlrWPZTAqRUk5TU1cKp8goJVhMSId6TvV5Ovdy1Z3nOAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f124bdb382c62ad871e40b79ef7527d3372c5c73a4df3203046638be0ebe2b06","last_reissued_at":"2026-07-05T08:57:24.022588Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:57:24.022588Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"JPEG-LM: LLMs as Image Generators with Canonical Codec Representations","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Marjan Ghazvininejad, Pang Wei Koh, Xiaochuang Han, Yulia Tsvetkov","submitted_at":"2024-08-15T23:57:02Z","abstract_excerpt":"Recent work in image and video generation has been adopting the autoregressive LLM architecture due to its generality and potentially easy integration into multi-modal systems. The crux of applying autoregressive training in language generation to visual generation is discretization -- representing continuous data like images and videos as discrete tokens. Common methods of discretizing images and videos include modeling raw pixel values, which are prohibitively lengthy, or vector quantization, which requires convoluted pre-hoc training. In this work, we propose to directly model images and vi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.08459","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.08459/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.08459","created_at":"2026-07-05T08:57:24.022656+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.08459v2","created_at":"2026-07-05T08:57:24.022656+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.08459","created_at":"2026-07-05T08:57:24.022656+00:00"},{"alias_kind":"pith_short_12","alias_value":"6ESL3M4CYYVN","created_at":"2026-07-05T08:57:24.022656+00:00"},{"alias_kind":"pith_short_16","alias_value":"6ESL3M4CYYVNQ4PE","created_at":"2026-07-05T08:57:24.022656+00:00"},{"alias_kind":"pith_short_8","alias_value":"6ESL3M4C","created_at":"2026-07-05T08:57:24.022656+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.04736","citing_title":"ChipSeek: Optimizing Verilog Generation via EDA-Integrated Reinforcement Learning","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6ESL3M4CYYVNQ4PEBN4665JH2M","json":"https://pith.science/pith/6ESL3M4CYYVNQ4PEBN4665JH2M.json","graph_json":"https://pith.science/api/pith-number/6ESL3M4CYYVNQ4PEBN4665JH2M/graph.json","events_json":"https://pith.science/api/pith-number/6ESL3M4CYYVNQ4PEBN4665JH2M/events.json","paper":"https://pith.science/paper/6ESL3M4C"},"agent_actions":{"view_html":"https://pith.science/pith/6ESL3M4CYYVNQ4PEBN4665JH2M","download_json":"https://pith.science/pith/6ESL3M4CYYVNQ4PEBN4665JH2M.json","view_paper":"https://pith.science/paper/6ESL3M4C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.08459&json=true","fetch_graph":"https://pith.science/api/pith-number/6ESL3M4CYYVNQ4PEBN4665JH2M/graph.json","fetch_events":"https://pith.science/api/pith-number/6ESL3M4CYYVNQ4PEBN4665JH2M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6ESL3M4CYYVNQ4PEBN4665JH2M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6ESL3M4CYYVNQ4PEBN4665JH2M/action/storage_attestation","attest_author":"https://pith.science/pith/6ESL3M4CYYVNQ4PEBN4665JH2M/action/author_attestation","sign_citation":"https://pith.science/pith/6ESL3M4CYYVNQ4PEBN4665JH2M/action/citation_signature","submit_replication":"https://pith.science/pith/6ESL3M4CYYVNQ4PEBN4665JH2M/action/replication_record"}},"created_at":"2026-07-05T08:57:24.022656+00:00","updated_at":"2026-07-05T08:57:24.022656+00:00"}