{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V47V7GXSOIXQHZS6BRRNLBYQRE","short_pith_number":"pith:V47V7GXS","schema_version":"1.0","canonical_sha256":"af3f5f9af2722f03e65e0c62d58710892d77853a78b85ec3e4c4bb0a380c2529","source":{"kind":"arxiv","id":"2402.01093","version":2},"attestation_state":"computed","paper":{"title":"Need a Small Specialized Language Model? Plan Early!","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Angelos Katharopoulos, Awni Hannun, David Grangier, Pierre Ablin","submitted_at":"2024-02-02T01:45:18Z","abstract_excerpt":"Large language models are versatile tools but are not suitable for small inference budgets. Small models have more efficient inference, but their lower capacity means that their performance can be good only if one limits their scope to a specialized domain. This paper explores how to get good specialized small language models using a large, generic, pretraining set and a limited amount of specialized data. We consider two scenarios, depending on whether (i) one can afford pretraining a model for each specialization task, or (ii) one wants to cheaply adapt a single pretrained model for each tas"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.01093","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-02T01:45:18Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"77a938b9c1b3b18a614f0a76858e252ebb9f0cc88be6e68dc3f5d56d6abaf825","abstract_canon_sha256":"cd143598992f737dcbe4ebcbd01fab9313ae82ee110251ecc2fcd993e19a9166"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:29:13.779902Z","signature_b64":"V25gCDG15F8qs0pv4Zhe64+Z06LgZBQ0vAnAAht2xs9uSEGxK8/2aJHgfU5miEnYX5ypdbj31MzeVhfX8UVEDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af3f5f9af2722f03e65e0c62d58710892d77853a78b85ec3e4c4bb0a380c2529","last_reissued_at":"2026-07-05T09:29:13.779402Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:29:13.779402Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Need a Small Specialized Language Model? Plan Early!","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Angelos Katharopoulos, Awni Hannun, David Grangier, Pierre Ablin","submitted_at":"2024-02-02T01:45:18Z","abstract_excerpt":"Large language models are versatile tools but are not suitable for small inference budgets. Small models have more efficient inference, but their lower capacity means that their performance can be good only if one limits their scope to a specialized domain. This paper explores how to get good specialized small language models using a large, generic, pretraining set and a limited amount of specialized data. We consider two scenarios, depending on whether (i) one can afford pretraining a model for each specialization task, or (ii) one wants to cheaply adapt a single pretrained model for each tas"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.01093","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.01093/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.01093","created_at":"2026-07-05T09:29:13.779461+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.01093v2","created_at":"2026-07-05T09:29:13.779461+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.01093","created_at":"2026-07-05T09:29:13.779461+00:00"},{"alias_kind":"pith_short_12","alias_value":"V47V7GXSOIXQ","created_at":"2026-07-05T09:29:13.779461+00:00"},{"alias_kind":"pith_short_16","alias_value":"V47V7GXSOIXQHZS6","created_at":"2026-07-05T09:29:13.779461+00:00"},{"alias_kind":"pith_short_8","alias_value":"V47V7GXS","created_at":"2026-07-05T09:29:13.779461+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.12966","citing_title":"Assessing the Role of Data Quality in Training Bilingual Language Models","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V47V7GXSOIXQHZS6BRRNLBYQRE","json":"https://pith.science/pith/V47V7GXSOIXQHZS6BRRNLBYQRE.json","graph_json":"https://pith.science/api/pith-number/V47V7GXSOIXQHZS6BRRNLBYQRE/graph.json","events_json":"https://pith.science/api/pith-number/V47V7GXSOIXQHZS6BRRNLBYQRE/events.json","paper":"https://pith.science/paper/V47V7GXS"},"agent_actions":{"view_html":"https://pith.science/pith/V47V7GXSOIXQHZS6BRRNLBYQRE","download_json":"https://pith.science/pith/V47V7GXSOIXQHZS6BRRNLBYQRE.json","view_paper":"https://pith.science/paper/V47V7GXS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.01093&json=true","fetch_graph":"https://pith.science/api/pith-number/V47V7GXSOIXQHZS6BRRNLBYQRE/graph.json","fetch_events":"https://pith.science/api/pith-number/V47V7GXSOIXQHZS6BRRNLBYQRE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V47V7GXSOIXQHZS6BRRNLBYQRE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V47V7GXSOIXQHZS6BRRNLBYQRE/action/storage_attestation","attest_author":"https://pith.science/pith/V47V7GXSOIXQHZS6BRRNLBYQRE/action/author_attestation","sign_citation":"https://pith.science/pith/V47V7GXSOIXQHZS6BRRNLBYQRE/action/citation_signature","submit_replication":"https://pith.science/pith/V47V7GXSOIXQHZS6BRRNLBYQRE/action/replication_record"}},"created_at":"2026-07-05T09:29:13.779461+00:00","updated_at":"2026-07-05T09:29:13.779461+00:00"}