{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ULPSHE5BM4ZHYXOLNXHZ2FIT7V","short_pith_number":"pith:ULPSHE5B","schema_version":"1.0","canonical_sha256":"a2df2393a167327c5dcb6dcf9d1513fd7a48cf3a6fc6dcb16357acbef9e4f2fb","source":{"kind":"arxiv","id":"2311.15436","version":1},"attestation_state":"computed","paper":{"title":"Learning to Skip for Language Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Claire Cui, Dewen Zeng, Nan Du, Tao Lei, Tao Wang, Yuanzhong Xu, Zhifeng Chen","submitted_at":"2023-11-26T21:45:53Z","abstract_excerpt":"Overparameterized large-scale language models have impressive generalization performance of in-context few-shot learning. However, most language models allocate the same amount of parameters or computation to each token, disregarding the complexity or importance of the input data. We argue that in language model pretraining, a variable amount of computation should be assigned to different tokens, and this can be efficiently achieved via a simple routing mechanism. Different from conventional early stopping techniques where tokens can early exit at only early layers, we propose a more general m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.15436","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-26T21:45:53Z","cross_cats_sorted":[],"title_canon_sha256":"db42b5b1e7f7c91ba9f5c28adc57f12e9e807cd23e2dd3193a0c4feb92a11596","abstract_canon_sha256":"553ef0ff569b5418f8abe0933aaaf0f986987a8cc7d7c1723790f14434f5c332"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:17:03.779238Z","signature_b64":"ZC2TLf4b7R8534JXBOtCW85w+QXoVeU7XpOtDxTP+RZ4PP13MYTg3Ppd1cXXmIa/9npoWFkYnzz6eMx3PejwDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a2df2393a167327c5dcb6dcf9d1513fd7a48cf3a6fc6dcb16357acbef9e4f2fb","last_reissued_at":"2026-07-05T07:17:03.778843Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:17:03.778843Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Skip for Language Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Claire Cui, Dewen Zeng, Nan Du, Tao Lei, Tao Wang, Yuanzhong Xu, Zhifeng Chen","submitted_at":"2023-11-26T21:45:53Z","abstract_excerpt":"Overparameterized large-scale language models have impressive generalization performance of in-context few-shot learning. However, most language models allocate the same amount of parameters or computation to each token, disregarding the complexity or importance of the input data. We argue that in language model pretraining, a variable amount of computation should be assigned to different tokens, and this can be efficiently achieved via a simple routing mechanism. Different from conventional early stopping techniques where tokens can early exit at only early layers, we propose a more general m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.15436","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.15436/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.15436","created_at":"2026-07-05T07:17:03.778897+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.15436v1","created_at":"2026-07-05T07:17:03.778897+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.15436","created_at":"2026-07-05T07:17:03.778897+00:00"},{"alias_kind":"pith_short_12","alias_value":"ULPSHE5BM4ZH","created_at":"2026-07-05T07:17:03.778897+00:00"},{"alias_kind":"pith_short_16","alias_value":"ULPSHE5BM4ZHYXOL","created_at":"2026-07-05T07:17:03.778897+00:00"},{"alias_kind":"pith_short_8","alias_value":"ULPSHE5B","created_at":"2026-07-05T07:17:03.778897+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27743","citing_title":"End-to-End Dynamic Sparsity for Resource-Adaptive LLM Inference","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ULPSHE5BM4ZHYXOLNXHZ2FIT7V","json":"https://pith.science/pith/ULPSHE5BM4ZHYXOLNXHZ2FIT7V.json","graph_json":"https://pith.science/api/pith-number/ULPSHE5BM4ZHYXOLNXHZ2FIT7V/graph.json","events_json":"https://pith.science/api/pith-number/ULPSHE5BM4ZHYXOLNXHZ2FIT7V/events.json","paper":"https://pith.science/paper/ULPSHE5B"},"agent_actions":{"view_html":"https://pith.science/pith/ULPSHE5BM4ZHYXOLNXHZ2FIT7V","download_json":"https://pith.science/pith/ULPSHE5BM4ZHYXOLNXHZ2FIT7V.json","view_paper":"https://pith.science/paper/ULPSHE5B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.15436&json=true","fetch_graph":"https://pith.science/api/pith-number/ULPSHE5BM4ZHYXOLNXHZ2FIT7V/graph.json","fetch_events":"https://pith.science/api/pith-number/ULPSHE5BM4ZHYXOLNXHZ2FIT7V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ULPSHE5BM4ZHYXOLNXHZ2FIT7V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ULPSHE5BM4ZHYXOLNXHZ2FIT7V/action/storage_attestation","attest_author":"https://pith.science/pith/ULPSHE5BM4ZHYXOLNXHZ2FIT7V/action/author_attestation","sign_citation":"https://pith.science/pith/ULPSHE5BM4ZHYXOLNXHZ2FIT7V/action/citation_signature","submit_replication":"https://pith.science/pith/ULPSHE5BM4ZHYXOLNXHZ2FIT7V/action/replication_record"}},"created_at":"2026-07-05T07:17:03.778897+00:00","updated_at":"2026-07-05T07:17:03.778897+00:00"}