{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4I66OPSKL52DFZX3UHAGNH6BKY","short_pith_number":"pith:4I66OPSK","schema_version":"1.0","canonical_sha256":"e23de73e4a5f7432e6fba1c0669fc15630bba6955591fe17add6813429790e7a","source":{"kind":"arxiv","id":"2408.17377","version":1},"attestation_state":"computed","paper":{"title":"NDP: Next Distribution Prediction as a More Broad Target","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Abudukeyumu Abudula, Bei Li, Chenglong Wang, Jingbo Zhu, Junhao Ruan, Tong Xiao, Xinyu Liu, Yinqiao Li, Yuan Ge, Yuchun Fan","submitted_at":"2024-08-30T16:13:49Z","abstract_excerpt":"Large language models (LLMs) trained on next-token prediction (NTP) paradigm have demonstrated powerful capabilities. However, the existing NTP paradigm contains several limitations, particularly related to planned task complications and error propagation during inference. In our work, we extend the critique of NTP, highlighting its limitation also due to training with a narrow objective: the prediction of a sub-optimal one-hot distribution. To support this critique, we conducted a pre-experiment treating the output distribution from powerful LLMs as efficient world data compression. By evalua"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.17377","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-08-30T16:13:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cf922f8c9456f260869fcc703db5c7653e223f3a24a2aea5d43c4683cf31c7d9","abstract_canon_sha256":"f520ae0c89e965a2be04275a8042cdb3152cc22d2bcd0da78d0d668ac96d1b5b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:01:14.641150Z","signature_b64":"1T0/K4YIvkZ8txM42TbvTJsu+8J9i6iKd9L+QbmuuA12qgm6bs0IEKGxReoRyobvaNb6DW5ao0b5UzZRO/CVCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e23de73e4a5f7432e6fba1c0669fc15630bba6955591fe17add6813429790e7a","last_reissued_at":"2026-07-05T09:01:14.640679Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:01:14.640679Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NDP: Next Distribution Prediction as a More Broad Target","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Abudukeyumu Abudula, Bei Li, Chenglong Wang, Jingbo Zhu, Junhao Ruan, Tong Xiao, Xinyu Liu, Yinqiao Li, Yuan Ge, Yuchun Fan","submitted_at":"2024-08-30T16:13:49Z","abstract_excerpt":"Large language models (LLMs) trained on next-token prediction (NTP) paradigm have demonstrated powerful capabilities. However, the existing NTP paradigm contains several limitations, particularly related to planned task complications and error propagation during inference. In our work, we extend the critique of NTP, highlighting its limitation also due to training with a narrow objective: the prediction of a sub-optimal one-hot distribution. To support this critique, we conducted a pre-experiment treating the output distribution from powerful LLMs as efficient world data compression. By evalua"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.17377","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.17377/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.17377","created_at":"2026-07-05T09:01:14.640734+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.17377v1","created_at":"2026-07-05T09:01:14.640734+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.17377","created_at":"2026-07-05T09:01:14.640734+00:00"},{"alias_kind":"pith_short_12","alias_value":"4I66OPSKL52D","created_at":"2026-07-05T09:01:14.640734+00:00"},{"alias_kind":"pith_short_16","alias_value":"4I66OPSKL52DFZX3","created_at":"2026-07-05T09:01:14.640734+00:00"},{"alias_kind":"pith_short_8","alias_value":"4I66OPSK","created_at":"2026-07-05T09:01:14.640734+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.18572","citing_title":"Back into Plato's Cave: Examining Cross-modal Representational Convergence at Scale","ref_index":77,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4I66OPSKL52DFZX3UHAGNH6BKY","json":"https://pith.science/pith/4I66OPSKL52DFZX3UHAGNH6BKY.json","graph_json":"https://pith.science/api/pith-number/4I66OPSKL52DFZX3UHAGNH6BKY/graph.json","events_json":"https://pith.science/api/pith-number/4I66OPSKL52DFZX3UHAGNH6BKY/events.json","paper":"https://pith.science/paper/4I66OPSK"},"agent_actions":{"view_html":"https://pith.science/pith/4I66OPSKL52DFZX3UHAGNH6BKY","download_json":"https://pith.science/pith/4I66OPSKL52DFZX3UHAGNH6BKY.json","view_paper":"https://pith.science/paper/4I66OPSK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.17377&json=true","fetch_graph":"https://pith.science/api/pith-number/4I66OPSKL52DFZX3UHAGNH6BKY/graph.json","fetch_events":"https://pith.science/api/pith-number/4I66OPSKL52DFZX3UHAGNH6BKY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4I66OPSKL52DFZX3UHAGNH6BKY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4I66OPSKL52DFZX3UHAGNH6BKY/action/storage_attestation","attest_author":"https://pith.science/pith/4I66OPSKL52DFZX3UHAGNH6BKY/action/author_attestation","sign_citation":"https://pith.science/pith/4I66OPSKL52DFZX3UHAGNH6BKY/action/citation_signature","submit_replication":"https://pith.science/pith/4I66OPSKL52DFZX3UHAGNH6BKY/action/replication_record"}},"created_at":"2026-07-05T09:01:14.640734+00:00","updated_at":"2026-07-05T09:01:14.640734+00:00"}