{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:Y5FUN5MMSMGPB6WZYBRU5FK6YB","short_pith_number":"pith:Y5FUN5MM","schema_version":"1.0","canonical_sha256":"c74b46f58c930cf0fad9c0634e955ec04f092d3177042fa2b5f714c63535409f","source":{"kind":"arxiv","id":"2210.13459","version":1},"attestation_state":"computed","paper":{"title":"Adaptive Label Smoothing with Self-Knowledge in Natural Language Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Dongkyu Lee, Ka Chun Cheung, Nevin L. Zhang","submitted_at":"2022-10-22T11:52:38Z","abstract_excerpt":"Overconfidence has been shown to impair generalization and calibration of a neural network. Previous studies remedy this issue by adding a regularization term to a loss function, preventing a model from making a peaked distribution. Label smoothing smoothes target labels with a pre-defined prior label distribution; as a result, a model is learned to maximize the likelihood of predicting the soft label. Nonetheless, the amount of smoothing is the same in all samples and remains fixed in training. In other words, label smoothing does not reflect the change in probability distribution mapped by a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.13459","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-22T11:52:38Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"c73102f153fdd2f9dc0228a20075f3c3d1d83abc0c5b06722d6712c99e552de4","abstract_canon_sha256":"eb2ea9ddd262a64e4b96828bc665fc86c66f8fe6a860b2b967e0651b21fbf1be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:10:18.669118Z","signature_b64":"3Sj/9QUvmMK8iKmSvF+bTiufNpB122nHzuyiFTRrfPX3fmOpE33T8P7G0ZtCkTbFjIFMB4+lBzuMG5ewXG0gCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c74b46f58c930cf0fad9c0634e955ec04f092d3177042fa2b5f714c63535409f","last_reissued_at":"2026-07-05T05:10:18.668635Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:10:18.668635Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Label Smoothing with Self-Knowledge in Natural Language Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Dongkyu Lee, Ka Chun Cheung, Nevin L. Zhang","submitted_at":"2022-10-22T11:52:38Z","abstract_excerpt":"Overconfidence has been shown to impair generalization and calibration of a neural network. Previous studies remedy this issue by adding a regularization term to a loss function, preventing a model from making a peaked distribution. Label smoothing smoothes target labels with a pre-defined prior label distribution; as a result, a model is learned to maximize the likelihood of predicting the soft label. Nonetheless, the amount of smoothing is the same in all samples and remains fixed in training. In other words, label smoothing does not reflect the change in probability distribution mapped by a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.13459","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.13459/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.13459","created_at":"2026-07-05T05:10:18.668703+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.13459v1","created_at":"2026-07-05T05:10:18.668703+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.13459","created_at":"2026-07-05T05:10:18.668703+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y5FUN5MMSMGP","created_at":"2026-07-05T05:10:18.668703+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y5FUN5MMSMGPB6WZ","created_at":"2026-07-05T05:10:18.668703+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y5FUN5MM","created_at":"2026-07-05T05:10:18.668703+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.00602","citing_title":"Mitigating Heterogeneous Token Overfitting in LLM Knowledge Editing","ref_index":25,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y5FUN5MMSMGPB6WZYBRU5FK6YB","json":"https://pith.science/pith/Y5FUN5MMSMGPB6WZYBRU5FK6YB.json","graph_json":"https://pith.science/api/pith-number/Y5FUN5MMSMGPB6WZYBRU5FK6YB/graph.json","events_json":"https://pith.science/api/pith-number/Y5FUN5MMSMGPB6WZYBRU5FK6YB/events.json","paper":"https://pith.science/paper/Y5FUN5MM"},"agent_actions":{"view_html":"https://pith.science/pith/Y5FUN5MMSMGPB6WZYBRU5FK6YB","download_json":"https://pith.science/pith/Y5FUN5MMSMGPB6WZYBRU5FK6YB.json","view_paper":"https://pith.science/paper/Y5FUN5MM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.13459&json=true","fetch_graph":"https://pith.science/api/pith-number/Y5FUN5MMSMGPB6WZYBRU5FK6YB/graph.json","fetch_events":"https://pith.science/api/pith-number/Y5FUN5MMSMGPB6WZYBRU5FK6YB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y5FUN5MMSMGPB6WZYBRU5FK6YB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y5FUN5MMSMGPB6WZYBRU5FK6YB/action/storage_attestation","attest_author":"https://pith.science/pith/Y5FUN5MMSMGPB6WZYBRU5FK6YB/action/author_attestation","sign_citation":"https://pith.science/pith/Y5FUN5MMSMGPB6WZYBRU5FK6YB/action/citation_signature","submit_replication":"https://pith.science/pith/Y5FUN5MMSMGPB6WZYBRU5FK6YB/action/replication_record"}},"created_at":"2026-07-05T05:10:18.668703+00:00","updated_at":"2026-07-05T05:10:18.668703+00:00"}