{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2016:VL3WXSYJ2FP34EL2QPKVVPYFNS","short_pith_number":"pith:VL3WXSYJ","schema_version":"1.0","canonical_sha256":"aaf76bcb09d15fbe117a83d55abf056cb9df80b525c95ee008319469da7d0539","source":{"kind":"arxiv","id":"1606.08415","version":5},"attestation_state":"computed","paper":{"title":"Gaussian Error Linear Units (GELUs)","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"The GELU activation xΦ(x) outperforms ReLU and ELU on computer vision, natural language processing, and speech tasks.","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dan Hendrycks, Kevin Gimpel","submitted_at":"2016-06-27T19:20:40Z","abstract_excerpt":"We propose the Gaussian Error Linear Unit (GELU), a high-performing neural network activation function. The GELU activation function is $x\\Phi(x)$, where $\\Phi(x)$ the standard Gaussian cumulative distribution function. The GELU nonlinearity weights inputs by their value, rather than gates inputs by their sign as in ReLUs ($x\\mathbf{1}_{x>0}$). We perform an empirical evaluation of the GELU nonlinearity against the ReLU and ELU activations and find performance improvements across all considered computer vision, natural language processing, and speech tasks."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":true,"formal_links_present":true},"canonical_record":{"source":{"id":"1606.08415","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2016-06-27T19:20:40Z","cross_cats_sorted":[],"title_canon_sha256":"a5ca717da6d88c1aba68c34fa663664c69d6046b7244dcf0f983783d2d9824e9","abstract_canon_sha256":"d19ddc86b6f6af403cdb4ea404a16d8076ee7f391816001e4b01a7e10ab54c37"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:17:34.982267Z","signature_b64":"zP/G84yNmTEuD2dQZZh0kRMjoZjfvsJgeP+FJ4Ufs1tiC9PT/EaBRSrT77foruLIYL/oX/NAboiowkcMzIi8AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aaf76bcb09d15fbe117a83d55abf056cb9df80b525c95ee008319469da7d0539","last_reissued_at":"2026-07-05T06:17:34.981584Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:17:34.981584Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Gaussian Error Linear Units (GELUs)","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"The GELU activation xΦ(x) outperforms ReLU and ELU on computer vision, natural language processing, and speech tasks.","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dan Hendrycks, Kevin Gimpel","submitted_at":"2016-06-27T19:20:40Z","abstract_excerpt":"We propose the Gaussian Error Linear Unit (GELU), a high-performing neural network activation function. The GELU activation function is $x\\Phi(x)$, where $\\Phi(x)$ the standard Gaussian cumulative distribution function. The GELU nonlinearity weights inputs by their value, rather than gates inputs by their sign as in ReLUs ($x\\mathbf{1}_{x>0}$). We perform an empirical evaluation of the GELU nonlinearity against the ReLU and ELU activations and find performance improvements across all considered computer vision, natural language processing, and speech tasks."},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"We perform an empirical evaluation of the GELU nonlinearity against the ReLU and ELU activations and find performance improvements across all considered computer vision, natural language processing, and speech tasks.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the observed gains on the specific tasks and models tested will generalize to other architectures, datasets, and training regimes without further tuning.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"GELU activation xΦ(x) outperforms ReLU and ELU on computer vision, NLP, and speech tasks by weighting inputs by value rather than gating by sign.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"The GELU activation xΦ(x) outperforms ReLU and ELU on computer vision, natural language processing, and speech tasks.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"b83c2b8b1ed504cbde17bb7710d5cbb0ff0e9d2c4eb1d879968aa3d31e368a13"},"source":{"id":"1606.08415","kind":"arxiv","version":5},"verdict":{"id":"7a5657e6-2360-44ec-b2a4-020dcf342710","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T12:12:07.972046Z","strongest_claim":"We perform an empirical evaluation of the GELU nonlinearity against the ReLU and ELU activations and find performance improvements across all considered computer vision, natural language processing, and speech tasks.","one_line_summary":"GELU activation xΦ(x) outperforms ReLU and ELU on computer vision, NLP, and speech tasks by weighting inputs by value rather than gating by sign.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the observed gains on the specific tasks and models tested will generalize to other architectures, datasets, and training regimes without further tuning.","pith_extraction_headline":"The GELU activation xΦ(x) outperforms ReLU and ELU on computer vision, natural language processing, and speech tasks."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1606.08415/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":50,"sample":[{"doi":"","year":null,"title":"Adaptive dropout for training deep neural networks , year =","work_id":"826fb3d9-a3e6-4eaf-8754-cdde878e5a0b","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"Learning with pseudo-ensembles , year =","work_id":"5686e296-a575-4d9c-b51e-ca063964598c","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"International Conference on Learning Representations , title =","work_id":"af8adfb7-c005-46f0-a112-dc4457e4627e","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"A Simple Approximation to the Area Under Standard Normal Curve , year =","work_id":"5fdadacd-9a9c-446c-bede-d35471e9e8f4","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"Natural Neural Networks , year =","work_id":"517a3f62-3ec1-4077-9263-5464291823a4","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":50,"snapshot_sha256":"181f21b5bd42e8b39f59516aa9d95e98a8e4cca8b1a14087d794a3f29c3cbf43","internal_anchors":0},"formal_canon":{"evidence_count":2,"snapshot_sha256":"9f3f89fe5fe484d3df62e5d29159aa44c63eedb310a386bdfed1b199af3b5c83"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1606.08415","created_at":"2026-07-05T06:17:34.981800+00:00"},{"alias_kind":"arxiv_version","alias_value":"1606.08415v5","created_at":"2026-07-05T06:17:34.981800+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1606.08415","created_at":"2026-07-05T06:17:34.981800+00:00"},{"alias_kind":"pith_short_12","alias_value":"VL3WXSYJ2FP3","created_at":"2026-07-05T06:17:34.981800+00:00"},{"alias_kind":"pith_short_16","alias_value":"VL3WXSYJ2FP34EL2","created_at":"2026-07-05T06:17:34.981800+00:00"},{"alias_kind":"pith_short_8","alias_value":"VL3WXSYJ","created_at":"2026-07-05T06:17:34.981800+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":391,"internal_anchor_count":364,"sample":[{"citing_arxiv_id":"2607.05736","citing_title":"Multimodal Molecular Representation Learning with Graph Neural Networks, Deep & Cross Networks, and SMILES Embeddings","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08323","citing_title":"Intrinsic Instantaneous Coarse-to-Fine Recoverability in the Lorenz-96 System","ref_index":40,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08044","citing_title":"Catching Disguised Transients with ASTRANet: Anomaly-Aware Spectroscopic Classification and Conformal Calibration","ref_index":81,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05803","citing_title":"Quantifying and Expanding the Theoretical Capacity of Late-Interaction Retrieval Models","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06002","citing_title":"Solving Hamiltonian Constraint Equation with Physics-Informed Neural Networks","ref_index":54,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05196","citing_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","ref_index":72,"is_internal_anchor":true},{"citing_arxiv_id":"2604.18483","citing_title":"Steadily moving semi-infinite fracture in plane poroelasticity","ref_index":33,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25811","citing_title":"Hierarchical Graph Learning for Calendar Spread Strategies in Commodity Futures Markets","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25915","citing_title":"FunPiQ: A New Benchmark for Pixel-Level Quality Assessment in Fundus Images","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25596","citing_title":"Slay the Shear: A Unified Statistical Framework for Weak Gravitational Lensing Shear Estimation","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25519","citing_title":"Quantization Inflates Reasoning: Token Inflation as a Hidden Cost of Low-Bit Reasoning Models","ref_index":149,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25318","citing_title":"REViT: Roto-reflection Equivariant Convolutional Vision Transformer","ref_index":56,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24752","citing_title":"Can Scale Save Us From Plasticity Loss in Large Language Models?","ref_index":69,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24851","citing_title":"Real vs. Complex Spectral Bases for Neural Operators: The Role of Green's Function Alignment","ref_index":33,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24396","citing_title":"Parallel Manifold Steering: Efficient Adaptation of Large Associative Memories via Residual Energy Shaping","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24375","citing_title":"MATCH: Flow Matching for Multi-View Anomaly Detection","ref_index":37,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24184","citing_title":"Project Ariadne: Prompt-Conditioned Route Generation for Synthesis Planning","ref_index":61,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26416","citing_title":"Methane-Plume Segmentation From Hyperspectral Satellite Imagery Via Multimodal Deep Learning","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24072","citing_title":"Fabric Image Demoir\\'eing Benchmark from Synthesis to Restoration","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26913","citing_title":"Neural Texture Compression using Hypernetworks","ref_index":35,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26520","citing_title":"Multipath Adaptive Gated Bottleneck Latent ODE with Raman Data Fusion for Cell Culture Process Forecasting","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23346","citing_title":"Field-level weak lensing cosmology with $<100$ simulations using multifidelity simulation-based inference","ref_index":70,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23942","citing_title":"DREG: A Layer-Wise Jacobian Regularization as a General-Purpose Penalty","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21910","citing_title":"Fidelity- and Perception-Aware Local Implicit Attention for Arbitrary-Scale Image Super-Resolution","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21398","citing_title":"BIT-Nav: Brain-Inspired Trajectory Memory for Embodied Navigation","ref_index":35,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":2,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VL3WXSYJ2FP34EL2QPKVVPYFNS","json":"https://pith.science/pith/VL3WXSYJ2FP34EL2QPKVVPYFNS.json","graph_json":"https://pith.science/api/pith-number/VL3WXSYJ2FP34EL2QPKVVPYFNS/graph.json","events_json":"https://pith.science/api/pith-number/VL3WXSYJ2FP34EL2QPKVVPYFNS/events.json","paper":"https://pith.science/paper/VL3WXSYJ"},"agent_actions":{"view_html":"https://pith.science/pith/VL3WXSYJ2FP34EL2QPKVVPYFNS","download_json":"https://pith.science/pith/VL3WXSYJ2FP34EL2QPKVVPYFNS.json","view_paper":"https://pith.science/paper/VL3WXSYJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1606.08415&json=true","fetch_graph":"https://pith.science/api/pith-number/VL3WXSYJ2FP34EL2QPKVVPYFNS/graph.json","fetch_events":"https://pith.science/api/pith-number/VL3WXSYJ2FP34EL2QPKVVPYFNS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VL3WXSYJ2FP34EL2QPKVVPYFNS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VL3WXSYJ2FP34EL2QPKVVPYFNS/action/storage_attestation","attest_author":"https://pith.science/pith/VL3WXSYJ2FP34EL2QPKVVPYFNS/action/author_attestation","sign_citation":"https://pith.science/pith/VL3WXSYJ2FP34EL2QPKVVPYFNS/action/citation_signature","submit_replication":"https://pith.science/pith/VL3WXSYJ2FP34EL2QPKVVPYFNS/action/replication_record"}},"created_at":"2026-07-05T06:17:34.981800+00:00","updated_at":"2026-07-05T06:17:34.981800+00:00"}