{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:67I7MYR5IQZZSUMPGSLSBWIBHN","short_pith_number":"pith:67I7MYR5","schema_version":"1.0","canonical_sha256":"f7d1f6623d443399518f349720d9013b567711e113fc99b564afb207fd072af1","source":{"kind":"arxiv","id":"1908.09355","version":1},"attestation_state":"computed","paper":{"title":"Patient Knowledge Distillation for BERT Model Compression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jingjing Liu, Siqi Sun, Yu Cheng, Zhe Gan","submitted_at":"2019-08-25T16:13:24Z","abstract_excerpt":"Pre-trained language models such as BERT have proven to be highly effective for natural language processing (NLP) tasks. However, the high demand for computing resources in training such models hinders their application in practice. In order to alleviate this resource hunger in large-scale model training, we propose a Patient Knowledge Distillation approach to compress an original large model (teacher) into an equally-effective lightweight shallow network (student). Different from previous knowledge distillation methods, which only use the output from the last layer of the teacher network for "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.09355","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-08-25T16:13:24Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"cdf876163c7f0d041b4058d62bbfcc6e1b78281bd7b312a4dcb5de9087d1daa8","abstract_canon_sha256":"6e7b1e6881aca8b0dd1153aeffe379964b8452343e882fe340abe1622a29a048"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T23:59:39.416641Z","signature_b64":"O4mFqya+L6nu5ck7AnoOOd1InwSubgmPwP7bFrizm6HinC+RlvvOHEVfPbVO/qCG0MtOr2NwKzQwzsOFTXnLBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f7d1f6623d443399518f349720d9013b567711e113fc99b564afb207fd072af1","last_reissued_at":"2026-07-04T23:59:39.416163Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T23:59:39.416163Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Patient Knowledge Distillation for BERT Model Compression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jingjing Liu, Siqi Sun, Yu Cheng, Zhe Gan","submitted_at":"2019-08-25T16:13:24Z","abstract_excerpt":"Pre-trained language models such as BERT have proven to be highly effective for natural language processing (NLP) tasks. However, the high demand for computing resources in training such models hinders their application in practice. In order to alleviate this resource hunger in large-scale model training, we propose a Patient Knowledge Distillation approach to compress an original large model (teacher) into an equally-effective lightweight shallow network (student). Different from previous knowledge distillation methods, which only use the output from the last layer of the teacher network for "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.09355","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.09355/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.09355","created_at":"2026-07-04T23:59:39.416222+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.09355v1","created_at":"2026-07-04T23:59:39.416222+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.09355","created_at":"2026-07-04T23:59:39.416222+00:00"},{"alias_kind":"pith_short_12","alias_value":"67I7MYR5IQZZ","created_at":"2026-07-04T23:59:39.416222+00:00"},{"alias_kind":"pith_short_16","alias_value":"67I7MYR5IQZZSUMP","created_at":"2026-07-04T23:59:39.416222+00:00"},{"alias_kind":"pith_short_8","alias_value":"67I7MYR5","created_at":"2026-07-04T23:59:39.416222+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06796","citing_title":"Enhancing deep learning models for time series classification via knowledge distillation","ref_index":37,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24747","citing_title":"Scaling Laws for Task-Specific LLM Distillation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24956","citing_title":"NITP: Next Implicit Token Prediction for LLM Pre-training","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24956","citing_title":"NITP: Next Implicit Token Prediction for LLM Pre-training","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09881","citing_title":"Toward Calibrated, Fair, and accurate Deepfake Detection","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2508.03949","citing_title":"Model Compression vs. Adversarial Robustness: An Empirical Study on Language Models for Code","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2511.05476","citing_title":"A Metamorphic Testing Perspective on Knowledge Distillation for Language Models of Code: Does the Student Deeply Mimic the Teacher?","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2101.02388","citing_title":"Knowledge Distillation in Iterative Generative Models for Improved Sampling Speed","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"1909.11942","citing_title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11260","citing_title":"Curriculum Learning-Guided Progressive Distillation in Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24380","citing_title":"Structural Pruning of Large Vision Language Models: A Comprehensive Study on Pruning Dynamics, Recovery, and Data Efficiency","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00329","citing_title":"Fast Text-to-Audio Generation with One-Step Sampling via Energy-Scoring and Auxiliary Contextual Representation Distillation","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/67I7MYR5IQZZSUMPGSLSBWIBHN","json":"https://pith.science/pith/67I7MYR5IQZZSUMPGSLSBWIBHN.json","graph_json":"https://pith.science/api/pith-number/67I7MYR5IQZZSUMPGSLSBWIBHN/graph.json","events_json":"https://pith.science/api/pith-number/67I7MYR5IQZZSUMPGSLSBWIBHN/events.json","paper":"https://pith.science/paper/67I7MYR5"},"agent_actions":{"view_html":"https://pith.science/pith/67I7MYR5IQZZSUMPGSLSBWIBHN","download_json":"https://pith.science/pith/67I7MYR5IQZZSUMPGSLSBWIBHN.json","view_paper":"https://pith.science/paper/67I7MYR5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.09355&json=true","fetch_graph":"https://pith.science/api/pith-number/67I7MYR5IQZZSUMPGSLSBWIBHN/graph.json","fetch_events":"https://pith.science/api/pith-number/67I7MYR5IQZZSUMPGSLSBWIBHN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/67I7MYR5IQZZSUMPGSLSBWIBHN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/67I7MYR5IQZZSUMPGSLSBWIBHN/action/storage_attestation","attest_author":"https://pith.science/pith/67I7MYR5IQZZSUMPGSLSBWIBHN/action/author_attestation","sign_citation":"https://pith.science/pith/67I7MYR5IQZZSUMPGSLSBWIBHN/action/citation_signature","submit_replication":"https://pith.science/pith/67I7MYR5IQZZSUMPGSLSBWIBHN/action/replication_record"}},"created_at":"2026-07-04T23:59:39.416222+00:00","updated_at":"2026-07-04T23:59:39.416222+00:00"}