{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:PQMHZSOBODRJIPGK2FCPC7TBEP","short_pith_number":"pith:PQMHZSOB","schema_version":"1.0","canonical_sha256":"7c187cc9c170e2943ccad144f17e6123e6762901eaf3f2083493dfdf11020c4d","source":{"kind":"arxiv","id":"2004.12817","version":1},"attestation_state":"computed","paper":{"title":"LightPAFF: A Two-Stage Distillation Framework for Pre-training and Fine-tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hao Sun, Hongzhi Liu, Jianfeng Lu, Kaitao Song, Tao Qin, Tie-Yan Liu, Xu Tan","submitted_at":"2020-04-27T14:00:09Z","abstract_excerpt":"While pre-training and fine-tuning, e.g., BERT~\\citep{devlin2018bert}, GPT-2~\\citep{radford2019language}, have achieved great success in language understanding and generation tasks, the pre-trained models are usually too big for online deployment in terms of both memory cost and inference speed, which hinders them from practical online usage. In this paper, we propose LightPAFF, a Lightweight Pre-training And Fine-tuning Framework that leverages two-stage knowledge distillation to transfer knowledge from a big teacher model to a lightweight student model in both pre-training and fine-tuning st"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.12817","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-04-27T14:00:09Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"b02b63fa052653892cba0e1a3e8c19a4a54a9cbce8dbe67f58d23b360e11856b","abstract_canon_sha256":"422f5aad401556e8767af031e464ca38fc53792dc78712c9fc3b074c70ea05e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:58:27.434055Z","signature_b64":"hyVSnXKX2RtnqgghXt5B6Lbk0aBF0z0Rot0ccRyM6fPUtE0ijHg64+S7jGq1jjrrZTq1b+xAQ4UuMoW4SOv+AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c187cc9c170e2943ccad144f17e6123e6762901eaf3f2083493dfdf11020c4d","last_reissued_at":"2026-07-05T00:58:27.433612Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:58:27.433612Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LightPAFF: A Two-Stage Distillation Framework for Pre-training and Fine-tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hao Sun, Hongzhi Liu, Jianfeng Lu, Kaitao Song, Tao Qin, Tie-Yan Liu, Xu Tan","submitted_at":"2020-04-27T14:00:09Z","abstract_excerpt":"While pre-training and fine-tuning, e.g., BERT~\\citep{devlin2018bert}, GPT-2~\\citep{radford2019language}, have achieved great success in language understanding and generation tasks, the pre-trained models are usually too big for online deployment in terms of both memory cost and inference speed, which hinders them from practical online usage. In this paper, we propose LightPAFF, a Lightweight Pre-training And Fine-tuning Framework that leverages two-stage knowledge distillation to transfer knowledge from a big teacher model to a lightweight student model in both pre-training and fine-tuning st"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.12817","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.12817/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.12817","created_at":"2026-07-05T00:58:27.433670+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.12817v1","created_at":"2026-07-05T00:58:27.433670+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.12817","created_at":"2026-07-05T00:58:27.433670+00:00"},{"alias_kind":"pith_short_12","alias_value":"PQMHZSOBODRJ","created_at":"2026-07-05T00:58:27.433670+00:00"},{"alias_kind":"pith_short_16","alias_value":"PQMHZSOBODRJIPGK","created_at":"2026-07-05T00:58:27.433670+00:00"},{"alias_kind":"pith_short_8","alias_value":"PQMHZSOB","created_at":"2026-07-05T00:58:27.433670+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23124","citing_title":"PRIDE: Privileged Information-enhanced Distillation for Empathetic Dialogue Generation","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11260","citing_title":"Curriculum Learning-Guided Progressive Distillation in Large Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2306.08543","citing_title":"MiniLLM: On-Policy Distillation of Large Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20244","citing_title":"Hybrid Policy Distillation for LLMs","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PQMHZSOBODRJIPGK2FCPC7TBEP","json":"https://pith.science/pith/PQMHZSOBODRJIPGK2FCPC7TBEP.json","graph_json":"https://pith.science/api/pith-number/PQMHZSOBODRJIPGK2FCPC7TBEP/graph.json","events_json":"https://pith.science/api/pith-number/PQMHZSOBODRJIPGK2FCPC7TBEP/events.json","paper":"https://pith.science/paper/PQMHZSOB"},"agent_actions":{"view_html":"https://pith.science/pith/PQMHZSOBODRJIPGK2FCPC7TBEP","download_json":"https://pith.science/pith/PQMHZSOBODRJIPGK2FCPC7TBEP.json","view_paper":"https://pith.science/paper/PQMHZSOB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.12817&json=true","fetch_graph":"https://pith.science/api/pith-number/PQMHZSOBODRJIPGK2FCPC7TBEP/graph.json","fetch_events":"https://pith.science/api/pith-number/PQMHZSOBODRJIPGK2FCPC7TBEP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PQMHZSOBODRJIPGK2FCPC7TBEP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PQMHZSOBODRJIPGK2FCPC7TBEP/action/storage_attestation","attest_author":"https://pith.science/pith/PQMHZSOBODRJIPGK2FCPC7TBEP/action/author_attestation","sign_citation":"https://pith.science/pith/PQMHZSOBODRJIPGK2FCPC7TBEP/action/citation_signature","submit_replication":"https://pith.science/pith/PQMHZSOBODRJIPGK2FCPC7TBEP/action/replication_record"}},"created_at":"2026-07-05T00:58:27.433670+00:00","updated_at":"2026-07-05T00:58:27.433670+00:00"}