{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:NNR2M3BQAJTE3NIAM7HAMJYHTX","short_pith_number":"pith:NNR2M3BQ","schema_version":"1.0","canonical_sha256":"6b63a66c3002664db50067ce0627079dc45bb18538479d3c9f4c325e8fd1c8bd","source":{"kind":"arxiv","id":"1902.03393","version":2},"attestation_state":"computed","paper":{"title":"Improved Knowledge Distillation via Teacher Assistant","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Akihiro Matsukawa, Ang Li, Hassan Ghasemzadeh, Mehrdad Farajtabar, Nir Levine, Seyed-Iman Mirzadeh","submitted_at":"2019-02-09T09:06:01Z","abstract_excerpt":"Despite the fact that deep neural networks are powerful models and achieve appealing results on many tasks, they are too large to be deployed on edge devices like smartphones or embedded sensor nodes. There have been efforts to compress these networks, and a popular method is knowledge distillation, where a large (teacher) pre-trained network is used to train a smaller (student) network. However, in this paper, we show that the student network performance degrades when the gap between student and teacher is large. Given a fixed student network, one cannot employ an arbitrarily large teacher, o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1902.03393","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-02-09T09:06:01Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"db212afbe2b7a4a3c1ee075178987f26fcfbe431c0034c6b7e2cd12773c552c3","abstract_canon_sha256":"848169d40e6c1056d3b060069698bc2c55a734c734b97fd3ac72cc8924c43241"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:26:23.512934Z","signature_b64":"xa31VszzQLbmBCEUMkBoKaCxSN8Ld24va9D6AocCTSUe23qDj4AKxrrbUurqotmWVyCrAZddcreF0j+r99p7Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6b63a66c3002664db50067ce0627079dc45bb18538479d3c9f4c325e8fd1c8bd","last_reissued_at":"2026-07-05T00:26:23.512474Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:26:23.512474Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improved Knowledge Distillation via Teacher Assistant","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Akihiro Matsukawa, Ang Li, Hassan Ghasemzadeh, Mehrdad Farajtabar, Nir Levine, Seyed-Iman Mirzadeh","submitted_at":"2019-02-09T09:06:01Z","abstract_excerpt":"Despite the fact that deep neural networks are powerful models and achieve appealing results on many tasks, they are too large to be deployed on edge devices like smartphones or embedded sensor nodes. There have been efforts to compress these networks, and a popular method is knowledge distillation, where a large (teacher) pre-trained network is used to train a smaller (student) network. However, in this paper, we show that the student network performance degrades when the gap between student and teacher is large. Given a fixed student network, one cannot employ an arbitrarily large teacher, o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1902.03393","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1902.03393/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1902.03393","created_at":"2026-07-05T00:26:23.512532+00:00"},{"alias_kind":"arxiv_version","alias_value":"1902.03393v2","created_at":"2026-07-05T00:26:23.512532+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1902.03393","created_at":"2026-07-05T00:26:23.512532+00:00"},{"alias_kind":"pith_short_12","alias_value":"NNR2M3BQAJTE","created_at":"2026-07-05T00:26:23.512532+00:00"},{"alias_kind":"pith_short_16","alias_value":"NNR2M3BQAJTE3NIA","created_at":"2026-07-05T00:26:23.512532+00:00"},{"alias_kind":"pith_short_8","alias_value":"NNR2M3BQ","created_at":"2026-07-05T00:26:23.512532+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08268","citing_title":"Different Teachers, Different Capabilities: Sub-1B On-Device Distillation for Structured Text Enrichment","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21851","citing_title":"TALAS: Teacher-Anchored Layer Alignment with Adaptive Sharpness-Aware Minimization for Embedding Distillation","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31293","citing_title":"Divergence Decoding: Inference-Time Unlearning via Auxiliary Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29726","citing_title":"SLAD : Shared LoRA Adapters for Task Specific Distillation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"1907.02226","citing_title":"Graph-based Knowledge Distillation by Multi-head Attention Network","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NNR2M3BQAJTE3NIAM7HAMJYHTX","json":"https://pith.science/pith/NNR2M3BQAJTE3NIAM7HAMJYHTX.json","graph_json":"https://pith.science/api/pith-number/NNR2M3BQAJTE3NIAM7HAMJYHTX/graph.json","events_json":"https://pith.science/api/pith-number/NNR2M3BQAJTE3NIAM7HAMJYHTX/events.json","paper":"https://pith.science/paper/NNR2M3BQ"},"agent_actions":{"view_html":"https://pith.science/pith/NNR2M3BQAJTE3NIAM7HAMJYHTX","download_json":"https://pith.science/pith/NNR2M3BQAJTE3NIAM7HAMJYHTX.json","view_paper":"https://pith.science/paper/NNR2M3BQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1902.03393&json=true","fetch_graph":"https://pith.science/api/pith-number/NNR2M3BQAJTE3NIAM7HAMJYHTX/graph.json","fetch_events":"https://pith.science/api/pith-number/NNR2M3BQAJTE3NIAM7HAMJYHTX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NNR2M3BQAJTE3NIAM7HAMJYHTX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NNR2M3BQAJTE3NIAM7HAMJYHTX/action/storage_attestation","attest_author":"https://pith.science/pith/NNR2M3BQAJTE3NIAM7HAMJYHTX/action/author_attestation","sign_citation":"https://pith.science/pith/NNR2M3BQAJTE3NIAM7HAMJYHTX/action/citation_signature","submit_replication":"https://pith.science/pith/NNR2M3BQAJTE3NIAM7HAMJYHTX/action/replication_record"}},"created_at":"2026-07-05T00:26:23.512532+00:00","updated_at":"2026-07-05T00:26:23.512532+00:00"}