{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4E7KDX7B3FYOWFWFLFWCYEMEQZ","short_pith_number":"pith:4E7KDX7B","schema_version":"1.0","canonical_sha256":"e13ea1dfe1d970eb16c5596c2c1184865600ffd4b6b7a8a0168617bf50b60b80","source":{"kind":"arxiv","id":"2505.22757","version":1},"attestation_state":"computed","paper":{"title":"Pre-Training Curriculum for Multi-Token Prediction in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alan Akbik, Ansar Aynetdinov","submitted_at":"2025-05-28T18:19:18Z","abstract_excerpt":"Multi-token prediction (MTP) is a recently proposed pre-training objective for language models. Rather than predicting only the next token (NTP), MTP predicts the next $k$ tokens at each prediction step, using multiple prediction heads. MTP has shown promise in improving downstream performance, inference speed, and training efficiency, particularly for large models. However, prior work has shown that smaller language models (SLMs) struggle with the MTP objective. To address this, we propose a curriculum learning strategy for MTP training, exploring two variants: a forward curriculum, which gra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22757","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-28T18:19:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0fe18e7d32488eeac9d4aa8c4f9e9eefa89b879e985a9b09dec90ccc32d34a13","abstract_canon_sha256":"8c0855444788dca531c6987e125091c9e9b9ee414f6b8f246be7acd57e648479"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:49.667799Z","signature_b64":"sD3nNUhrWF/dw2smtUaSJQDDwk/ejRbHwY5StCIBLBmMLkReAhrzg5coz0/SrTcfPkQq12LvALhw+cYA6+xoAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e13ea1dfe1d970eb16c5596c2c1184865600ffd4b6b7a8a0168617bf50b60b80","last_reissued_at":"2026-07-05T11:11:49.667283Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:49.667283Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pre-Training Curriculum for Multi-Token Prediction in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alan Akbik, Ansar Aynetdinov","submitted_at":"2025-05-28T18:19:18Z","abstract_excerpt":"Multi-token prediction (MTP) is a recently proposed pre-training objective for language models. Rather than predicting only the next token (NTP), MTP predicts the next $k$ tokens at each prediction step, using multiple prediction heads. MTP has shown promise in improving downstream performance, inference speed, and training efficiency, particularly for large models. However, prior work has shown that smaller language models (SLMs) struggle with the MTP objective. To address this, we propose a curriculum learning strategy for MTP training, exploring two variants: a forward curriculum, which gra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22757","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22757/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22757","created_at":"2026-07-05T11:11:49.667352+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22757v1","created_at":"2026-07-05T11:11:49.667352+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22757","created_at":"2026-07-05T11:11:49.667352+00:00"},{"alias_kind":"pith_short_12","alias_value":"4E7KDX7B3FYO","created_at":"2026-07-05T11:11:49.667352+00:00"},{"alias_kind":"pith_short_16","alias_value":"4E7KDX7B3FYOWFWF","created_at":"2026-07-05T11:11:49.667352+00:00"},{"alias_kind":"pith_short_8","alias_value":"4E7KDX7B","created_at":"2026-07-05T11:11:49.667352+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4E7KDX7B3FYOWFWFLFWCYEMEQZ","json":"https://pith.science/pith/4E7KDX7B3FYOWFWFLFWCYEMEQZ.json","graph_json":"https://pith.science/api/pith-number/4E7KDX7B3FYOWFWFLFWCYEMEQZ/graph.json","events_json":"https://pith.science/api/pith-number/4E7KDX7B3FYOWFWFLFWCYEMEQZ/events.json","paper":"https://pith.science/paper/4E7KDX7B"},"agent_actions":{"view_html":"https://pith.science/pith/4E7KDX7B3FYOWFWFLFWCYEMEQZ","download_json":"https://pith.science/pith/4E7KDX7B3FYOWFWFLFWCYEMEQZ.json","view_paper":"https://pith.science/paper/4E7KDX7B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22757&json=true","fetch_graph":"https://pith.science/api/pith-number/4E7KDX7B3FYOWFWFLFWCYEMEQZ/graph.json","fetch_events":"https://pith.science/api/pith-number/4E7KDX7B3FYOWFWFLFWCYEMEQZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4E7KDX7B3FYOWFWFLFWCYEMEQZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4E7KDX7B3FYOWFWFLFWCYEMEQZ/action/storage_attestation","attest_author":"https://pith.science/pith/4E7KDX7B3FYOWFWFLFWCYEMEQZ/action/author_attestation","sign_citation":"https://pith.science/pith/4E7KDX7B3FYOWFWFLFWCYEMEQZ/action/citation_signature","submit_replication":"https://pith.science/pith/4E7KDX7B3FYOWFWFLFWCYEMEQZ/action/replication_record"}},"created_at":"2026-07-05T11:11:49.667352+00:00","updated_at":"2026-07-05T11:11:49.667352+00:00"}