{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:VSGBFURZPB65MFCRCIIRAN3EZA","short_pith_number":"pith:VSGBFURZ","schema_version":"1.0","canonical_sha256":"ac8c12d239787dd614511211103764c827c80c977ff83cf689833bfe586fb6c8","source":{"kind":"arxiv","id":"2004.12993","version":1},"attestation_state":"computed","paper":{"title":"DeeBERT: Dynamic Early Exiting for Accelerating BERT Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jaejun Lee, Jimmy Lin, Ji Xin, Raphael Tang, Yaoliang Yu","submitted_at":"2020-04-27T17:58:05Z","abstract_excerpt":"Large-scale pre-trained language models such as BERT have brought significant improvements to NLP applications. However, they are also notorious for being slow in inference, which makes them difficult to deploy in real-time applications. We propose a simple but effective method, DeeBERT, to accelerate BERT inference. Our approach allows samples to exit earlier without passing through the entire model. Experiments show that DeeBERT is able to save up to ~40% inference time with minimal degradation in model quality. Further analyses show different behaviors in the BERT transformer layers and als"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.12993","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-04-27T17:58:05Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1bbd1d03b192a4255bb12439d73db6d9410c00fcd3e07597e287abf41e934e8e","abstract_canon_sha256":"23e10bbf062d188fdb9f6ad92788bb2d3ddcc1e69ccae94a6fba1ca60af2b324"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:58:29.305727Z","signature_b64":"x+dB6t3D+Qu7KH9PfXyacQcmmDxaOaVpCrQqjlC7TiGZHKEl28u34NOXxawUJnPDVXetLgWUqiVdYcKT9xmWCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac8c12d239787dd614511211103764c827c80c977ff83cf689833bfe586fb6c8","last_reissued_at":"2026-07-05T00:58:29.305220Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:58:29.305220Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DeeBERT: Dynamic Early Exiting for Accelerating BERT Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jaejun Lee, Jimmy Lin, Ji Xin, Raphael Tang, Yaoliang Yu","submitted_at":"2020-04-27T17:58:05Z","abstract_excerpt":"Large-scale pre-trained language models such as BERT have brought significant improvements to NLP applications. However, they are also notorious for being slow in inference, which makes them difficult to deploy in real-time applications. We propose a simple but effective method, DeeBERT, to accelerate BERT inference. Our approach allows samples to exit earlier without passing through the entire model. Experiments show that DeeBERT is able to save up to ~40% inference time with minimal degradation in model quality. Further analyses show different behaviors in the BERT transformer layers and als"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.12993","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.12993/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.12993","created_at":"2026-07-05T00:58:29.305276+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.12993v1","created_at":"2026-07-05T00:58:29.305276+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.12993","created_at":"2026-07-05T00:58:29.305276+00:00"},{"alias_kind":"pith_short_12","alias_value":"VSGBFURZPB65","created_at":"2026-07-05T00:58:29.305276+00:00"},{"alias_kind":"pith_short_16","alias_value":"VSGBFURZPB65MFCR","created_at":"2026-07-05T00:58:29.305276+00:00"},{"alias_kind":"pith_short_8","alias_value":"VSGBFURZ","created_at":"2026-07-05T00:58:29.305276+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26538","citing_title":"CascadeFormer: Depth-Tapered Transformers Motivated by Gradient Fan-in Asymmetry","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09937","citing_title":"RKSC: Reasoning-Aware KV Cache Sharing and Confident Early Exit for Multi-Step LLM Inference","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06574","citing_title":"Skip a Layer or Loop It? Learning Program-of-Layers in LLMs","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09165","citing_title":"Sparse Layers are Critical to Scaling Looped Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24810","citing_title":"A Comparative Analysis on the Performance of Upper Confidence Bound Algorithms in Adaptive Deep Neural Networks","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2410.04941","citing_title":"TOAST: Transformer Optimization using Adaptive and Simple Transformations","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2510.12773","citing_title":"Dr.LLM: Dynamic Layer Routing in LLMs","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18592","citing_title":"Two-dimensional early exit optimisation of LLM inference","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2303.08112","citing_title":"Eliciting Latent Predictions from Transformers with the Tuned Lens","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09165","citing_title":"Sparse Layers are Critical to Scaling Looped Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10875","citing_title":"Compute Where it Counts: Self Optimizing Language Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24810","citing_title":"A Comparative Analysis on the Performance of Upper Confidence Bound Algorithms in Adaptive Deep Neural Networks","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22906","citing_title":"Network Edge Inference for Large Language Models: Principles, Techniques, and Opportunities","ref_index":174,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05741","citing_title":"HyperLens: Quantifying Cognitive Effort in LLMs with Fine-grained Confidence Trajectory","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02285","citing_title":"Complexity Horizons of Compressed Models in Analog Circuit Analysis","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VSGBFURZPB65MFCRCIIRAN3EZA","json":"https://pith.science/pith/VSGBFURZPB65MFCRCIIRAN3EZA.json","graph_json":"https://pith.science/api/pith-number/VSGBFURZPB65MFCRCIIRAN3EZA/graph.json","events_json":"https://pith.science/api/pith-number/VSGBFURZPB65MFCRCIIRAN3EZA/events.json","paper":"https://pith.science/paper/VSGBFURZ"},"agent_actions":{"view_html":"https://pith.science/pith/VSGBFURZPB65MFCRCIIRAN3EZA","download_json":"https://pith.science/pith/VSGBFURZPB65MFCRCIIRAN3EZA.json","view_paper":"https://pith.science/paper/VSGBFURZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.12993&json=true","fetch_graph":"https://pith.science/api/pith-number/VSGBFURZPB65MFCRCIIRAN3EZA/graph.json","fetch_events":"https://pith.science/api/pith-number/VSGBFURZPB65MFCRCIIRAN3EZA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VSGBFURZPB65MFCRCIIRAN3EZA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VSGBFURZPB65MFCRCIIRAN3EZA/action/storage_attestation","attest_author":"https://pith.science/pith/VSGBFURZPB65MFCRCIIRAN3EZA/action/author_attestation","sign_citation":"https://pith.science/pith/VSGBFURZPB65MFCRCIIRAN3EZA/action/citation_signature","submit_replication":"https://pith.science/pith/VSGBFURZPB65MFCRCIIRAN3EZA/action/replication_record"}},"created_at":"2026-07-05T00:58:29.305276+00:00","updated_at":"2026-07-05T00:58:29.305276+00:00"}