{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:BUKWETKALFP35ZC7QWX6C3H36D","short_pith_number":"pith:BUKWETKA","schema_version":"1.0","canonical_sha256":"0d15624d40595fbee45f85afe16cfbf0d81631a2b59ac97b9188caeedf321eee","source":{"kind":"arxiv","id":"1910.01769","version":2},"attestation_state":"computed","paper":{"title":"Distilling BERT into Simple Neural Networks with Unlabeled Transfer Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Ahmed Hassan Awadallah, Subhabrata Mukherjee","submitted_at":"2019-10-04T01:01:26Z","abstract_excerpt":"Recent advances in pre-training huge models on large amounts of text through self supervision have obtained state-of-the-art results in various natural language processing tasks. However, these huge and expensive models are difficult to use in practise for downstream tasks. Some recent efforts use knowledge distillation to compress these models. However, we see a gap between the performance of the smaller student models as compared to that of the large teacher. In this work, we leverage large amounts of in-domain unlabeled transfer data in addition to a limited amount of labeled training insta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.01769","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-10-04T01:01:26Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"e2e5ee398fdb87d1e78b4a959e32a79a7a703cbffa5eb1933592db7b44811d8b","abstract_canon_sha256":"1b117e8ed0e5e03c0ae6e47f007296ec88f7c4ba6d82872703631fcb7ae4d0d3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:21:34.859966Z","signature_b64":"Rqtp0Y47CYjt6w/FzKJD/r5FIrJvOp0JLcskAVFBuKjy1CGCDmF1xCMASOfS8lulIID3BDj52UFMPXDw3CdfCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0d15624d40595fbee45f85afe16cfbf0d81631a2b59ac97b9188caeedf321eee","last_reissued_at":"2026-07-05T01:21:34.859554Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:21:34.859554Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distilling BERT into Simple Neural Networks with Unlabeled Transfer Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Ahmed Hassan Awadallah, Subhabrata Mukherjee","submitted_at":"2019-10-04T01:01:26Z","abstract_excerpt":"Recent advances in pre-training huge models on large amounts of text through self supervision have obtained state-of-the-art results in various natural language processing tasks. However, these huge and expensive models are difficult to use in practise for downstream tasks. Some recent efforts use knowledge distillation to compress these models. However, we see a gap between the performance of the smaller student models as compared to that of the large teacher. In this work, we leverage large amounts of in-domain unlabeled transfer data in addition to a limited amount of labeled training insta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.01769","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.01769/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.01769","created_at":"2026-07-05T01:21:34.859617+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.01769v2","created_at":"2026-07-05T01:21:34.859617+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.01769","created_at":"2026-07-05T01:21:34.859617+00:00"},{"alias_kind":"pith_short_12","alias_value":"BUKWETKALFP3","created_at":"2026-07-05T01:21:34.859617+00:00"},{"alias_kind":"pith_short_16","alias_value":"BUKWETKALFP35ZC7","created_at":"2026-07-05T01:21:34.859617+00:00"},{"alias_kind":"pith_short_8","alias_value":"BUKWETKA","created_at":"2026-07-05T01:21:34.859617+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.17371","citing_title":"ADEPT: Architecture-Driven Energy-Efficient CNN Fine-Tuning on PIM Accelerators","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BUKWETKALFP35ZC7QWX6C3H36D","json":"https://pith.science/pith/BUKWETKALFP35ZC7QWX6C3H36D.json","graph_json":"https://pith.science/api/pith-number/BUKWETKALFP35ZC7QWX6C3H36D/graph.json","events_json":"https://pith.science/api/pith-number/BUKWETKALFP35ZC7QWX6C3H36D/events.json","paper":"https://pith.science/paper/BUKWETKA"},"agent_actions":{"view_html":"https://pith.science/pith/BUKWETKALFP35ZC7QWX6C3H36D","download_json":"https://pith.science/pith/BUKWETKALFP35ZC7QWX6C3H36D.json","view_paper":"https://pith.science/paper/BUKWETKA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.01769&json=true","fetch_graph":"https://pith.science/api/pith-number/BUKWETKALFP35ZC7QWX6C3H36D/graph.json","fetch_events":"https://pith.science/api/pith-number/BUKWETKALFP35ZC7QWX6C3H36D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BUKWETKALFP35ZC7QWX6C3H36D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BUKWETKALFP35ZC7QWX6C3H36D/action/storage_attestation","attest_author":"https://pith.science/pith/BUKWETKALFP35ZC7QWX6C3H36D/action/author_attestation","sign_citation":"https://pith.science/pith/BUKWETKALFP35ZC7QWX6C3H36D/action/citation_signature","submit_replication":"https://pith.science/pith/BUKWETKALFP35ZC7QWX6C3H36D/action/replication_record"}},"created_at":"2026-07-05T01:21:34.859617+00:00","updated_at":"2026-07-05T01:21:34.859617+00:00"}