{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YL2YPG66MYVTV7SID7VJBFZNFZ","short_pith_number":"pith:YL2YPG66","schema_version":"1.0","canonical_sha256":"c2f5879bde662b3afe481fea90972d2e63c00b013dbfffddf683c69393efd741","source":{"kind":"arxiv","id":"2404.07647","version":1},"attestation_state":"computed","paper":{"title":"Why do small language models underperform? Studying Language Model Saturation via the Softmax Bottleneck","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Beno\\^it Sagot, \\'Eric de la Clergerie, Nathan Godey","submitted_at":"2024-04-11T11:10:36Z","abstract_excerpt":"Recent advances in language modeling consist in pretraining highly parameterized neural networks on extremely large web-mined text corpora. Training and inference with such models can be costly in practice, which incentivizes the use of smaller counterparts. However, it has been observed that smaller models can suffer from saturation, characterized as a drop in performance at some advanced point in training followed by a plateau. In this paper, we find that such saturation can be explained by a mismatch between the hidden dimension of smaller models and the high rank of the target contextual p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.07647","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-11T11:10:36Z","cross_cats_sorted":[],"title_canon_sha256":"def5209b67fdb90d8bd00ff2204abdc3cbe36cb9a627c126ff72bdbb0e97ee8e","abstract_canon_sha256":"41c42a14c5f378f3eafe25a399fb272cdb061bb3a6b68f129c630aa1e1d8978f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:06:55.732880Z","signature_b64":"V5n+gM5Ddq++mnfoKBYX/8OwtBG1wjDRN/hyeM69rme1T8PiNSYBSLnFiMtqarQ7BshzJ7OhFSYNi4p3j8skDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c2f5879bde662b3afe481fea90972d2e63c00b013dbfffddf683c69393efd741","last_reissued_at":"2026-07-05T08:06:55.732394Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:06:55.732394Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Why do small language models underperform? Studying Language Model Saturation via the Softmax Bottleneck","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Beno\\^it Sagot, \\'Eric de la Clergerie, Nathan Godey","submitted_at":"2024-04-11T11:10:36Z","abstract_excerpt":"Recent advances in language modeling consist in pretraining highly parameterized neural networks on extremely large web-mined text corpora. Training and inference with such models can be costly in practice, which incentivizes the use of smaller counterparts. However, it has been observed that smaller models can suffer from saturation, characterized as a drop in performance at some advanced point in training followed by a plateau. In this paper, we find that such saturation can be explained by a mismatch between the hidden dimension of smaller models and the high rank of the target contextual p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.07647","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.07647/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.07647","created_at":"2026-07-05T08:06:55.732454+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.07647v1","created_at":"2026-07-05T08:06:55.732454+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.07647","created_at":"2026-07-05T08:06:55.732454+00:00"},{"alias_kind":"pith_short_12","alias_value":"YL2YPG66MYVT","created_at":"2026-07-05T08:06:55.732454+00:00"},{"alias_kind":"pith_short_16","alias_value":"YL2YPG66MYVTV7SI","created_at":"2026-07-05T08:06:55.732454+00:00"},{"alias_kind":"pith_short_8","alias_value":"YL2YPG66","created_at":"2026-07-05T08:06:55.732454+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.09789","citing_title":"When Less is More: The LLM Scaling Paradox in Context Compression","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YL2YPG66MYVTV7SID7VJBFZNFZ","json":"https://pith.science/pith/YL2YPG66MYVTV7SID7VJBFZNFZ.json","graph_json":"https://pith.science/api/pith-number/YL2YPG66MYVTV7SID7VJBFZNFZ/graph.json","events_json":"https://pith.science/api/pith-number/YL2YPG66MYVTV7SID7VJBFZNFZ/events.json","paper":"https://pith.science/paper/YL2YPG66"},"agent_actions":{"view_html":"https://pith.science/pith/YL2YPG66MYVTV7SID7VJBFZNFZ","download_json":"https://pith.science/pith/YL2YPG66MYVTV7SID7VJBFZNFZ.json","view_paper":"https://pith.science/paper/YL2YPG66","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.07647&json=true","fetch_graph":"https://pith.science/api/pith-number/YL2YPG66MYVTV7SID7VJBFZNFZ/graph.json","fetch_events":"https://pith.science/api/pith-number/YL2YPG66MYVTV7SID7VJBFZNFZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YL2YPG66MYVTV7SID7VJBFZNFZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YL2YPG66MYVTV7SID7VJBFZNFZ/action/storage_attestation","attest_author":"https://pith.science/pith/YL2YPG66MYVTV7SID7VJBFZNFZ/action/author_attestation","sign_citation":"https://pith.science/pith/YL2YPG66MYVTV7SID7VJBFZNFZ/action/citation_signature","submit_replication":"https://pith.science/pith/YL2YPG66MYVTV7SID7VJBFZNFZ/action/replication_record"}},"created_at":"2026-07-05T08:06:55.732454+00:00","updated_at":"2026-07-05T08:06:55.732454+00:00"}