{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OC4PEHPWOBCH66J3ME6XSLYZL3","short_pith_number":"pith:OC4PEHPW","schema_version":"1.0","canonical_sha256":"70b8f21df670447f793b613d792f195ee32fe0bb0c78099110a7c7f2a0c3b3ce","source":{"kind":"arxiv","id":"2410.05078","version":2},"attestation_state":"computed","paper":{"title":"Compression via Pre-trained Transformers: A Study on Byte-Level Multimodal Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Anian Ruoss, David Heurtel-Depeiges, Joel Veness, Tim Genewein","submitted_at":"2024-10-07T14:32:03Z","abstract_excerpt":"Foundation models are strong data compressors, but when accounting for their parameter size, their compression ratios are inferior to standard compression algorithms. Naively reducing the parameter count does not necessarily help as it deteriorates predictions and, accordingly, compression. We conduct a large-scale empirical study to find a sweet spot where pre-trained vanilla transformers can achieve competitive compression ratios. To this end, we train models on 165GB of raw byte sequences of either text, image, or audio data (and all possible combinations of the three) and then compress 1GB"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.05078","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-07T14:32:03Z","cross_cats_sorted":["cs.AI","cs.IT","math.IT"],"title_canon_sha256":"534287d34ac3e2be0a4d424a5cc50574175c6f09bb908340ebfb832e92257aa1","abstract_canon_sha256":"c2f636ee77d77ff6612a6deb5c3f8c244c3f48fd5d267ac86778507e50033e5c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:01.528682Z","signature_b64":"os1bvFEFP3cotYMJ2FpeP52/EYApkCi7sdQpufjtPz7Tt3IikrSOV2QlAmmTJ6lT/Dvl3gh/W/Juki1VkhY1DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"70b8f21df670447f793b613d792f195ee32fe0bb0c78099110a7c7f2a0c3b3ce","last_reissued_at":"2026-07-05T11:08:01.528011Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:01.528011Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Compression via Pre-trained Transformers: A Study on Byte-Level Multimodal Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Anian Ruoss, David Heurtel-Depeiges, Joel Veness, Tim Genewein","submitted_at":"2024-10-07T14:32:03Z","abstract_excerpt":"Foundation models are strong data compressors, but when accounting for their parameter size, their compression ratios are inferior to standard compression algorithms. Naively reducing the parameter count does not necessarily help as it deteriorates predictions and, accordingly, compression. We conduct a large-scale empirical study to find a sweet spot where pre-trained vanilla transformers can achieve competitive compression ratios. To this end, we train models on 165GB of raw byte sequences of either text, image, or audio data (and all possible combinations of the three) and then compress 1GB"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.05078","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.05078/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.05078","created_at":"2026-07-05T11:08:01.528146+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.05078v2","created_at":"2026-07-05T11:08:01.528146+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.05078","created_at":"2026-07-05T11:08:01.528146+00:00"},{"alias_kind":"pith_short_12","alias_value":"OC4PEHPWOBCH","created_at":"2026-07-05T11:08:01.528146+00:00"},{"alias_kind":"pith_short_16","alias_value":"OC4PEHPWOBCH66J3","created_at":"2026-07-05T11:08:01.528146+00:00"},{"alias_kind":"pith_short_8","alias_value":"OC4PEHPW","created_at":"2026-07-05T11:08:01.528146+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08221","citing_title":"LUMI: Tokenizer-Agnostic LLM-Based Lossless Image Compression","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OC4PEHPWOBCH66J3ME6XSLYZL3","json":"https://pith.science/pith/OC4PEHPWOBCH66J3ME6XSLYZL3.json","graph_json":"https://pith.science/api/pith-number/OC4PEHPWOBCH66J3ME6XSLYZL3/graph.json","events_json":"https://pith.science/api/pith-number/OC4PEHPWOBCH66J3ME6XSLYZL3/events.json","paper":"https://pith.science/paper/OC4PEHPW"},"agent_actions":{"view_html":"https://pith.science/pith/OC4PEHPWOBCH66J3ME6XSLYZL3","download_json":"https://pith.science/pith/OC4PEHPWOBCH66J3ME6XSLYZL3.json","view_paper":"https://pith.science/paper/OC4PEHPW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.05078&json=true","fetch_graph":"https://pith.science/api/pith-number/OC4PEHPWOBCH66J3ME6XSLYZL3/graph.json","fetch_events":"https://pith.science/api/pith-number/OC4PEHPWOBCH66J3ME6XSLYZL3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OC4PEHPWOBCH66J3ME6XSLYZL3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OC4PEHPWOBCH66J3ME6XSLYZL3/action/storage_attestation","attest_author":"https://pith.science/pith/OC4PEHPWOBCH66J3ME6XSLYZL3/action/author_attestation","sign_citation":"https://pith.science/pith/OC4PEHPWOBCH66J3ME6XSLYZL3/action/citation_signature","submit_replication":"https://pith.science/pith/OC4PEHPWOBCH66J3ME6XSLYZL3/action/replication_record"}},"created_at":"2026-07-05T11:08:01.528146+00:00","updated_at":"2026-07-05T11:08:01.528146+00:00"}