{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:JIDL7XBV4FOBFSOMJGRLZJOE7Y","short_pith_number":"pith:JIDL7XBV","schema_version":"1.0","canonical_sha256":"4a06bfdc35e15c12c9cc49a2bca5c4fe3b200d1e0d674633d5d9264c3da43ba9","source":{"kind":"arxiv","id":"2307.08941","version":3},"attestation_state":"computed","paper":{"title":"MLP Fusion: Towards Efficient Fine-tuning of Dense and Mixture-of-Experts Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jingrui He, Mengting Ai, Tianxin Wei, Yifan Chen, Zeming Guo","submitted_at":"2023-07-18T03:12:51Z","abstract_excerpt":"Fine-tuning a pre-trained language model (PLM) emerges as the predominant strategy in many natural language processing applications. However, this process is known to be expensive, especially on edge devices with low computing power. While general approaches (e.g. quantization and distillation) have been widely studied to reduce the compute/memory of PLM fine-tuning, one-shot compression techniques specifically designed for fine-tuning remain largely unexplored. In this paper, we investigate the neural tangent kernel (NTK)--which reveals the gradient descent dynamics of neural networks--of the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.08941","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-07-18T03:12:51Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"e4aabd74d5a29b2c5ebea31ebfbdbd11a24944eb531ee2f4709d6a1393b8f6db","abstract_canon_sha256":"21fb8aed7790454a5f2b48c5712f70b66f881983ea586105523607219c66e411"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:56:47.461300Z","signature_b64":"Xd0K0SYfhqRVxCPWJCTvZOHl5ZOn/YGWf1w0lbvizZE+igyBQfi1dhykdEBOCPOL5ahQCmquBA7IDZHAkFKUCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4a06bfdc35e15c12c9cc49a2bca5c4fe3b200d1e0d674633d5d9264c3da43ba9","last_reissued_at":"2026-07-05T09:56:47.460818Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:56:47.460818Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MLP Fusion: Towards Efficient Fine-tuning of Dense and Mixture-of-Experts Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jingrui He, Mengting Ai, Tianxin Wei, Yifan Chen, Zeming Guo","submitted_at":"2023-07-18T03:12:51Z","abstract_excerpt":"Fine-tuning a pre-trained language model (PLM) emerges as the predominant strategy in many natural language processing applications. However, this process is known to be expensive, especially on edge devices with low computing power. While general approaches (e.g. quantization and distillation) have been widely studied to reduce the compute/memory of PLM fine-tuning, one-shot compression techniques specifically designed for fine-tuning remain largely unexplored. In this paper, we investigate the neural tangent kernel (NTK)--which reveals the gradient descent dynamics of neural networks--of the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.08941","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.08941/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.08941","created_at":"2026-07-05T09:56:47.460875+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.08941v3","created_at":"2026-07-05T09:56:47.460875+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.08941","created_at":"2026-07-05T09:56:47.460875+00:00"},{"alias_kind":"pith_short_12","alias_value":"JIDL7XBV4FOB","created_at":"2026-07-05T09:56:47.460875+00:00"},{"alias_kind":"pith_short_16","alias_value":"JIDL7XBV4FOBFSOM","created_at":"2026-07-05T09:56:47.460875+00:00"},{"alias_kind":"pith_short_8","alias_value":"JIDL7XBV","created_at":"2026-07-05T09:56:47.460875+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JIDL7XBV4FOBFSOMJGRLZJOE7Y","json":"https://pith.science/pith/JIDL7XBV4FOBFSOMJGRLZJOE7Y.json","graph_json":"https://pith.science/api/pith-number/JIDL7XBV4FOBFSOMJGRLZJOE7Y/graph.json","events_json":"https://pith.science/api/pith-number/JIDL7XBV4FOBFSOMJGRLZJOE7Y/events.json","paper":"https://pith.science/paper/JIDL7XBV"},"agent_actions":{"view_html":"https://pith.science/pith/JIDL7XBV4FOBFSOMJGRLZJOE7Y","download_json":"https://pith.science/pith/JIDL7XBV4FOBFSOMJGRLZJOE7Y.json","view_paper":"https://pith.science/paper/JIDL7XBV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.08941&json=true","fetch_graph":"https://pith.science/api/pith-number/JIDL7XBV4FOBFSOMJGRLZJOE7Y/graph.json","fetch_events":"https://pith.science/api/pith-number/JIDL7XBV4FOBFSOMJGRLZJOE7Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JIDL7XBV4FOBFSOMJGRLZJOE7Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JIDL7XBV4FOBFSOMJGRLZJOE7Y/action/storage_attestation","attest_author":"https://pith.science/pith/JIDL7XBV4FOBFSOMJGRLZJOE7Y/action/author_attestation","sign_citation":"https://pith.science/pith/JIDL7XBV4FOBFSOMJGRLZJOE7Y/action/citation_signature","submit_replication":"https://pith.science/pith/JIDL7XBV4FOBFSOMJGRLZJOE7Y/action/replication_record"}},"created_at":"2026-07-05T09:56:47.460875+00:00","updated_at":"2026-07-05T09:56:47.460875+00:00"}