{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:E32HR2XLWDPV3RVYRG4HEQMFG4","short_pith_number":"pith:E32HR2XL","schema_version":"1.0","canonical_sha256":"26f478eaebb0df5dc6b889b872418537354964b0dedc3bc5c6155eb61b207dca","source":{"kind":"arxiv","id":"2108.02170","version":1},"attestation_state":"computed","paper":{"title":"Curriculum learning for language modeling","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Daniel Campos","submitted_at":"2021-08-04T16:53:43Z","abstract_excerpt":"Language Models like ELMo and BERT have provided robust representations of natural language, which serve as the language understanding component for a diverse range of downstream tasks.Curriculum learning is a method that employs a structured training regime instead, which has been leveraged in computer vision and machine translation to improve model training speed and model performance. While language models have proven transformational for the natural language processing community, these models have proven expensive, energy-intensive, and challenging to train. In this work, we explore the ef"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.02170","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CL","submitted_at":"2021-08-04T16:53:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"505d7255218b07446b9408ea5518acc3fbf3efb8b8bd224447f33b6da07c6bbc","abstract_canon_sha256":"a08dcb1fe340c0a74640628f740547be69556040a2c646af3ed080c159d5c5a2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:03:22.288782Z","signature_b64":"Nr/vLoQ3NG/TeudpXSF6Jot7LGzkNUnPSppk1zfvLcN3dFAEKYcB+OwTUhuTJuYixo+tjUKQ3CwqoSn5WoYXDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"26f478eaebb0df5dc6b889b872418537354964b0dedc3bc5c6155eb61b207dca","last_reissued_at":"2026-07-05T03:03:22.288366Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:03:22.288366Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Curriculum learning for language modeling","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Daniel Campos","submitted_at":"2021-08-04T16:53:43Z","abstract_excerpt":"Language Models like ELMo and BERT have provided robust representations of natural language, which serve as the language understanding component for a diverse range of downstream tasks.Curriculum learning is a method that employs a structured training regime instead, which has been leveraged in computer vision and machine translation to improve model training speed and model performance. While language models have proven transformational for the natural language processing community, these models have proven expensive, energy-intensive, and challenging to train. In this work, we explore the ef"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.02170","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.02170/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.02170","created_at":"2026-07-05T03:03:22.288431+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.02170v1","created_at":"2026-07-05T03:03:22.288431+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.02170","created_at":"2026-07-05T03:03:22.288431+00:00"},{"alias_kind":"pith_short_12","alias_value":"E32HR2XLWDPV","created_at":"2026-07-05T03:03:22.288431+00:00"},{"alias_kind":"pith_short_16","alias_value":"E32HR2XLWDPV3RVY","created_at":"2026-07-05T03:03:22.288431+00:00"},{"alias_kind":"pith_short_8","alias_value":"E32HR2XL","created_at":"2026-07-05T03:03:22.288431+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.10237","citing_title":"The Benefits of Temporal Correlations: SGD Learns k-Juntas from Random Walks Efficiently","ref_index":82,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E32HR2XLWDPV3RVYRG4HEQMFG4","json":"https://pith.science/pith/E32HR2XLWDPV3RVYRG4HEQMFG4.json","graph_json":"https://pith.science/api/pith-number/E32HR2XLWDPV3RVYRG4HEQMFG4/graph.json","events_json":"https://pith.science/api/pith-number/E32HR2XLWDPV3RVYRG4HEQMFG4/events.json","paper":"https://pith.science/paper/E32HR2XL"},"agent_actions":{"view_html":"https://pith.science/pith/E32HR2XLWDPV3RVYRG4HEQMFG4","download_json":"https://pith.science/pith/E32HR2XLWDPV3RVYRG4HEQMFG4.json","view_paper":"https://pith.science/paper/E32HR2XL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.02170&json=true","fetch_graph":"https://pith.science/api/pith-number/E32HR2XLWDPV3RVYRG4HEQMFG4/graph.json","fetch_events":"https://pith.science/api/pith-number/E32HR2XLWDPV3RVYRG4HEQMFG4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E32HR2XLWDPV3RVYRG4HEQMFG4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E32HR2XLWDPV3RVYRG4HEQMFG4/action/storage_attestation","attest_author":"https://pith.science/pith/E32HR2XLWDPV3RVYRG4HEQMFG4/action/author_attestation","sign_citation":"https://pith.science/pith/E32HR2XLWDPV3RVYRG4HEQMFG4/action/citation_signature","submit_replication":"https://pith.science/pith/E32HR2XLWDPV3RVYRG4HEQMFG4/action/replication_record"}},"created_at":"2026-07-05T03:03:22.288431+00:00","updated_at":"2026-07-05T03:03:22.288431+00:00"}