{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BAA5CPNQ6IATJXJYGPXOC33L2T","short_pith_number":"pith:BAA5CPNQ","schema_version":"1.0","canonical_sha256":"0801d13db0f20134dd3833eee16f6bd4d7c88db72f8b3ae7903d5923d584370f","source":{"kind":"arxiv","id":"2404.03626","version":3},"attestation_state":"computed","paper":{"title":"Training LLMs over Neurally Compressed Text","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Adam Roberts, Alex Alemi, Brian Lester, Jaehoon Lee, Jascha Sohl-Dickstein, Jeffrey Pennington, Noah Constant","submitted_at":"2024-04-04T17:48:28Z","abstract_excerpt":"In this paper, we explore the idea of training large language models (LLMs) over highly compressed text. While standard subword tokenizers compress text by a small factor, neural text compressors can achieve much higher rates of compression. If it were possible to train LLMs directly over neurally compressed text, this would confer advantages in training and serving efficiency, as well as easier handling of long text spans. The main obstacle to this goal is that strong compression tends to produce opaque outputs that are not well-suited for learning. In particular, we find that text na\\\"ively "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.03626","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-04T17:48:28Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"856b69e14bb6d1bc77d9fd6a16be1b30089946c05ce816be93ef693b0be76851","abstract_canon_sha256":"1ae0bc1d65fc8999cdab2a19c82751e10ccb3431a33325537c06cf4156a59b6d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:48:32.265712Z","signature_b64":"xBb6pKMlT1csMycNCctuUUMGaFVa9JTvX2iwrAhrO9NYrnwTeeC3ImlYVWCDguX97runSvOGa5HWmrNQVkHyBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0801d13db0f20134dd3833eee16f6bd4d7c88db72f8b3ae7903d5923d584370f","last_reissued_at":"2026-07-05T09:48:32.265217Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:48:32.265217Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training LLMs over Neurally Compressed Text","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Adam Roberts, Alex Alemi, Brian Lester, Jaehoon Lee, Jascha Sohl-Dickstein, Jeffrey Pennington, Noah Constant","submitted_at":"2024-04-04T17:48:28Z","abstract_excerpt":"In this paper, we explore the idea of training large language models (LLMs) over highly compressed text. While standard subword tokenizers compress text by a small factor, neural text compressors can achieve much higher rates of compression. If it were possible to train LLMs directly over neurally compressed text, this would confer advantages in training and serving efficiency, as well as easier handling of long text spans. The main obstacle to this goal is that strong compression tends to produce opaque outputs that are not well-suited for learning. In particular, we find that text na\\\"ively "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.03626","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.03626/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.03626","created_at":"2026-07-05T09:48:32.265278+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.03626v3","created_at":"2026-07-05T09:48:32.265278+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.03626","created_at":"2026-07-05T09:48:32.265278+00:00"},{"alias_kind":"pith_short_12","alias_value":"BAA5CPNQ6IAT","created_at":"2026-07-05T09:48:32.265278+00:00"},{"alias_kind":"pith_short_16","alias_value":"BAA5CPNQ6IATJXJY","created_at":"2026-07-05T09:48:32.265278+00:00"},{"alias_kind":"pith_short_8","alias_value":"BAA5CPNQ","created_at":"2026-07-05T09:48:32.265278+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.04289","citing_title":"Proxy Compression for Language Modeling","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09630","citing_title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BAA5CPNQ6IATJXJYGPXOC33L2T","json":"https://pith.science/pith/BAA5CPNQ6IATJXJYGPXOC33L2T.json","graph_json":"https://pith.science/api/pith-number/BAA5CPNQ6IATJXJYGPXOC33L2T/graph.json","events_json":"https://pith.science/api/pith-number/BAA5CPNQ6IATJXJYGPXOC33L2T/events.json","paper":"https://pith.science/paper/BAA5CPNQ"},"agent_actions":{"view_html":"https://pith.science/pith/BAA5CPNQ6IATJXJYGPXOC33L2T","download_json":"https://pith.science/pith/BAA5CPNQ6IATJXJYGPXOC33L2T.json","view_paper":"https://pith.science/paper/BAA5CPNQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.03626&json=true","fetch_graph":"https://pith.science/api/pith-number/BAA5CPNQ6IATJXJYGPXOC33L2T/graph.json","fetch_events":"https://pith.science/api/pith-number/BAA5CPNQ6IATJXJYGPXOC33L2T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BAA5CPNQ6IATJXJYGPXOC33L2T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BAA5CPNQ6IATJXJYGPXOC33L2T/action/storage_attestation","attest_author":"https://pith.science/pith/BAA5CPNQ6IATJXJYGPXOC33L2T/action/author_attestation","sign_citation":"https://pith.science/pith/BAA5CPNQ6IATJXJYGPXOC33L2T/action/citation_signature","submit_replication":"https://pith.science/pith/BAA5CPNQ6IATJXJYGPXOC33L2T/action/replication_record"}},"created_at":"2026-07-05T09:48:32.265278+00:00","updated_at":"2026-07-05T09:48:32.265278+00:00"}