{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:O5CJU4SZ3T6KMP42ZQTBZOGMHB","short_pith_number":"pith:O5CJU4SZ","schema_version":"1.0","canonical_sha256":"77449a7259dcfca63f9acc261cb8cc3855b733a11642508eff93c2675a6c8822","source":{"kind":"arxiv","id":"1908.01851","version":1},"attestation_state":"computed","paper":{"title":"Self-Knowledge Distillation in Natural Language Processing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Heeyoul Choi, Sangchul Hahn","submitted_at":"2019-08-02T15:17:27Z","abstract_excerpt":"Since deep learning became a key player in natural language processing (NLP), many deep learning models have been showing remarkable performances in a variety of NLP tasks, and in some cases, they are even outperforming humans. Such high performance can be explained by efficient knowledge representation of deep learning models. While many methods have been proposed to learn more efficient representation, knowledge distillation from pretrained deep networks suggest that we can use more information from the soft target probability to train other neural networks. In this paper, we propose a new k"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.01851","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-08-02T15:17:27Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"093e2d4e774bdc25f6b30240b73494868187ace2893e4a6d2afe841c2824c9f0","abstract_canon_sha256":"9930da31358634db46b844ec5a8e242c18a5bb057551ec69b2b6e9ffbcb5ffb6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T23:51:48.360904Z","signature_b64":"mg0X/GUGATRtAJGil78G5ThHinI+TSTgYL0lEqDF17BXoKuFjTtFkMCQUvIUM3O6hlvWLKJhmh2NnjcppE7eBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77449a7259dcfca63f9acc261cb8cc3855b733a11642508eff93c2675a6c8822","last_reissued_at":"2026-07-04T23:51:48.360483Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T23:51:48.360483Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Knowledge Distillation in Natural Language Processing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Heeyoul Choi, Sangchul Hahn","submitted_at":"2019-08-02T15:17:27Z","abstract_excerpt":"Since deep learning became a key player in natural language processing (NLP), many deep learning models have been showing remarkable performances in a variety of NLP tasks, and in some cases, they are even outperforming humans. Such high performance can be explained by efficient knowledge representation of deep learning models. While many methods have been proposed to learn more efficient representation, knowledge distillation from pretrained deep networks suggest that we can use more information from the soft target probability to train other neural networks. In this paper, we propose a new k"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.01851","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.01851/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.01851","created_at":"2026-07-04T23:51:48.360541+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.01851v1","created_at":"2026-07-04T23:51:48.360541+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.01851","created_at":"2026-07-04T23:51:48.360541+00:00"},{"alias_kind":"pith_short_12","alias_value":"O5CJU4SZ3T6K","created_at":"2026-07-04T23:51:48.360541+00:00"},{"alias_kind":"pith_short_16","alias_value":"O5CJU4SZ3T6KMP42","created_at":"2026-07-04T23:51:48.360541+00:00"},{"alias_kind":"pith_short_8","alias_value":"O5CJU4SZ","created_at":"2026-07-04T23:51:48.360541+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.09608","citing_title":"Metric Learning with Progressive Self-Distillation for Audio-Visual Embedding Learning","ref_index":5,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O5CJU4SZ3T6KMP42ZQTBZOGMHB","json":"https://pith.science/pith/O5CJU4SZ3T6KMP42ZQTBZOGMHB.json","graph_json":"https://pith.science/api/pith-number/O5CJU4SZ3T6KMP42ZQTBZOGMHB/graph.json","events_json":"https://pith.science/api/pith-number/O5CJU4SZ3T6KMP42ZQTBZOGMHB/events.json","paper":"https://pith.science/paper/O5CJU4SZ"},"agent_actions":{"view_html":"https://pith.science/pith/O5CJU4SZ3T6KMP42ZQTBZOGMHB","download_json":"https://pith.science/pith/O5CJU4SZ3T6KMP42ZQTBZOGMHB.json","view_paper":"https://pith.science/paper/O5CJU4SZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.01851&json=true","fetch_graph":"https://pith.science/api/pith-number/O5CJU4SZ3T6KMP42ZQTBZOGMHB/graph.json","fetch_events":"https://pith.science/api/pith-number/O5CJU4SZ3T6KMP42ZQTBZOGMHB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O5CJU4SZ3T6KMP42ZQTBZOGMHB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O5CJU4SZ3T6KMP42ZQTBZOGMHB/action/storage_attestation","attest_author":"https://pith.science/pith/O5CJU4SZ3T6KMP42ZQTBZOGMHB/action/author_attestation","sign_citation":"https://pith.science/pith/O5CJU4SZ3T6KMP42ZQTBZOGMHB/action/citation_signature","submit_replication":"https://pith.science/pith/O5CJU4SZ3T6KMP42ZQTBZOGMHB/action/replication_record"}},"created_at":"2026-07-04T23:51:48.360541+00:00","updated_at":"2026-07-04T23:51:48.360541+00:00"}