{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:O37Y3GIKC3XFNTOOIW4DFJOXDS","short_pith_number":"pith:O37Y3GIK","schema_version":"1.0","canonical_sha256":"76ff8d990a16ee56cdce45b832a5d71cb5372ec9ebcd03426d163f7433943aa9","source":{"kind":"arxiv","id":"2010.13002","version":2},"attestation_state":"computed","paper":{"title":"Pre-trained Summarization Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alexander M. Rush, Sam Shleifer","submitted_at":"2020-10-24T23:15:43Z","abstract_excerpt":"Recent state-of-the-art approaches to summarization utilize large pre-trained Transformer models. Distilling these models to smaller student models has become critically important for practical use; however there are many different distillation methods proposed by the NLP literature. Recent work on distilling BERT for classification and regression tasks shows strong performance using direct knowledge distillation. Alternatively, machine translation practitioners distill using pseudo-labeling, where a small model is trained on the translations of a larger model. A third, simpler approach is to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.13002","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-10-24T23:15:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"efec17a7b6454ef5dfa4177349cdc557859ac34b56f39764c76435b9cfa96f9d","abstract_canon_sha256":"ce43e85fc46fe45a1615cd180605ca415c59c06a4158eafa1d93b64fdf09660b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:47:15.761081Z","signature_b64":"vV8PTnbT8VwpDdWxBIruzRLAf9Dnq/p3kKDSZGADxi35YfKr0EFEIBK/JkTkqyNOn9riBytctjIjjHgHEEOJDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"76ff8d990a16ee56cdce45b832a5d71cb5372ec9ebcd03426d163f7433943aa9","last_reissued_at":"2026-07-05T01:47:15.760648Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:47:15.760648Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pre-trained Summarization Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alexander M. Rush, Sam Shleifer","submitted_at":"2020-10-24T23:15:43Z","abstract_excerpt":"Recent state-of-the-art approaches to summarization utilize large pre-trained Transformer models. Distilling these models to smaller student models has become critically important for practical use; however there are many different distillation methods proposed by the NLP literature. Recent work on distilling BERT for classification and regression tasks shows strong performance using direct knowledge distillation. Alternatively, machine translation practitioners distill using pseudo-labeling, where a small model is trained on the translations of a larger model. A third, simpler approach is to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.13002","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.13002/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.13002","created_at":"2026-07-05T01:47:15.760705+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.13002v2","created_at":"2026-07-05T01:47:15.760705+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.13002","created_at":"2026-07-05T01:47:15.760705+00:00"},{"alias_kind":"pith_short_12","alias_value":"O37Y3GIKC3XF","created_at":"2026-07-05T01:47:15.760705+00:00"},{"alias_kind":"pith_short_16","alias_value":"O37Y3GIKC3XFNTOO","created_at":"2026-07-05T01:47:15.760705+00:00"},{"alias_kind":"pith_short_8","alias_value":"O37Y3GIK","created_at":"2026-07-05T01:47:15.760705+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24758","citing_title":"CANDLE: CTC-based Arabic Noisy-character Deduplication using a Lightweight Encoder","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20816","citing_title":"Don't Ignore the Tail: Decoupling top-K Probabilities for Efficient Language Model Distillation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12391","citing_title":"Chain-of-Models Pre-Training: Rethinking Training Acceleration of Vision Foundation Models","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O37Y3GIKC3XFNTOOIW4DFJOXDS","json":"https://pith.science/pith/O37Y3GIKC3XFNTOOIW4DFJOXDS.json","graph_json":"https://pith.science/api/pith-number/O37Y3GIKC3XFNTOOIW4DFJOXDS/graph.json","events_json":"https://pith.science/api/pith-number/O37Y3GIKC3XFNTOOIW4DFJOXDS/events.json","paper":"https://pith.science/paper/O37Y3GIK"},"agent_actions":{"view_html":"https://pith.science/pith/O37Y3GIKC3XFNTOOIW4DFJOXDS","download_json":"https://pith.science/pith/O37Y3GIKC3XFNTOOIW4DFJOXDS.json","view_paper":"https://pith.science/paper/O37Y3GIK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.13002&json=true","fetch_graph":"https://pith.science/api/pith-number/O37Y3GIKC3XFNTOOIW4DFJOXDS/graph.json","fetch_events":"https://pith.science/api/pith-number/O37Y3GIKC3XFNTOOIW4DFJOXDS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O37Y3GIKC3XFNTOOIW4DFJOXDS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O37Y3GIKC3XFNTOOIW4DFJOXDS/action/storage_attestation","attest_author":"https://pith.science/pith/O37Y3GIKC3XFNTOOIW4DFJOXDS/action/author_attestation","sign_citation":"https://pith.science/pith/O37Y3GIKC3XFNTOOIW4DFJOXDS/action/citation_signature","submit_replication":"https://pith.science/pith/O37Y3GIKC3XFNTOOIW4DFJOXDS/action/replication_record"}},"created_at":"2026-07-05T01:47:15.760705+00:00","updated_at":"2026-07-05T01:47:15.760705+00:00"}