{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:S42DLFKNLCBDMHAQ63TY3LMMJ6","short_pith_number":"pith:S42DLFKN","schema_version":"1.0","canonical_sha256":"973435954d5882361c10f6e78dad8c4f99c38570a087a699c9b37a2aa7511ae1","source":{"kind":"arxiv","id":"2110.02432","version":2},"attestation_state":"computed","paper":{"title":"KNOT: Knowledge Distillation using Optimal Transport for Solving NLP Tasks","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Rishabh Bhardwaj, Soujanya Poria, Tushar Vaidya","submitted_at":"2021-10-06T00:44:00Z","abstract_excerpt":"We propose a new approach, Knowledge Distillation using Optimal Transport (KNOT), to distill the natural language semantic knowledge from multiple teacher networks to a student network. KNOT aims to train a (global) student model by learning to minimize the optimal transport cost of its assigned probability distribution over the labels to the weighted sum of probabilities predicted by the (local) teacher models, under the constraints, that the student model does not have access to teacher models' parameters or training data. To evaluate the quality of knowledge transfer, we introduce a new met"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.02432","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2021-10-06T00:44:00Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e7c544859f36e66ad36998e90d02eb5d32c6180d3a80e0df96fca9a9b99e392f","abstract_canon_sha256":"1ecbb361d19355e616dc6990d868dd47d2ffdb6e1d50734abd9beaf4e50ad9c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:58:14.566672Z","signature_b64":"0k1KDEFt2/Drs8azfpukNMSMDysjKkoh6CS5M3TtHVAGiD9mB5FfJd4m6qo9nkj+rAWmSAofpYRk6dItOvLrBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"973435954d5882361c10f6e78dad8c4f99c38570a087a699c9b37a2aa7511ae1","last_reissued_at":"2026-07-05T04:58:14.566178Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:58:14.566178Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KNOT: Knowledge Distillation using Optimal Transport for Solving NLP Tasks","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Rishabh Bhardwaj, Soujanya Poria, Tushar Vaidya","submitted_at":"2021-10-06T00:44:00Z","abstract_excerpt":"We propose a new approach, Knowledge Distillation using Optimal Transport (KNOT), to distill the natural language semantic knowledge from multiple teacher networks to a student network. KNOT aims to train a (global) student model by learning to minimize the optimal transport cost of its assigned probability distribution over the labels to the weighted sum of probabilities predicted by the (local) teacher models, under the constraints, that the student model does not have access to teacher models' parameters or training data. To evaluate the quality of knowledge transfer, we introduce a new met"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.02432","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.02432/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.02432","created_at":"2026-07-05T04:58:14.566235+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.02432v2","created_at":"2026-07-05T04:58:14.566235+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.02432","created_at":"2026-07-05T04:58:14.566235+00:00"},{"alias_kind":"pith_short_12","alias_value":"S42DLFKNLCBD","created_at":"2026-07-05T04:58:14.566235+00:00"},{"alias_kind":"pith_short_16","alias_value":"S42DLFKNLCBDMHAQ","created_at":"2026-07-05T04:58:14.566235+00:00"},{"alias_kind":"pith_short_8","alias_value":"S42DLFKN","created_at":"2026-07-05T04:58:14.566235+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.14528","citing_title":"Multi-Level Optimal Transport for Universal Cross-Tokenizer Knowledge Distillation on Language Models","ref_index":6,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S42DLFKNLCBDMHAQ63TY3LMMJ6","json":"https://pith.science/pith/S42DLFKNLCBDMHAQ63TY3LMMJ6.json","graph_json":"https://pith.science/api/pith-number/S42DLFKNLCBDMHAQ63TY3LMMJ6/graph.json","events_json":"https://pith.science/api/pith-number/S42DLFKNLCBDMHAQ63TY3LMMJ6/events.json","paper":"https://pith.science/paper/S42DLFKN"},"agent_actions":{"view_html":"https://pith.science/pith/S42DLFKNLCBDMHAQ63TY3LMMJ6","download_json":"https://pith.science/pith/S42DLFKNLCBDMHAQ63TY3LMMJ6.json","view_paper":"https://pith.science/paper/S42DLFKN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.02432&json=true","fetch_graph":"https://pith.science/api/pith-number/S42DLFKNLCBDMHAQ63TY3LMMJ6/graph.json","fetch_events":"https://pith.science/api/pith-number/S42DLFKNLCBDMHAQ63TY3LMMJ6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S42DLFKNLCBDMHAQ63TY3LMMJ6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S42DLFKNLCBDMHAQ63TY3LMMJ6/action/storage_attestation","attest_author":"https://pith.science/pith/S42DLFKNLCBDMHAQ63TY3LMMJ6/action/author_attestation","sign_citation":"https://pith.science/pith/S42DLFKNLCBDMHAQ63TY3LMMJ6/action/citation_signature","submit_replication":"https://pith.science/pith/S42DLFKNLCBDMHAQ63TY3LMMJ6/action/replication_record"}},"created_at":"2026-07-05T04:58:14.566235+00:00","updated_at":"2026-07-05T04:58:14.566235+00:00"}