{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:2TJT5CKI7BCZGUUIFT4IZSHR5E","short_pith_number":"pith:2TJT5CKI","schema_version":"1.0","canonical_sha256":"d4d33e8948f8459352882cf88cc8f1e92a931d3ad5e3ab072e51f8d9966305d2","source":{"kind":"arxiv","id":"2009.02070","version":2},"attestation_state":"computed","paper":{"title":"AutoTrans: Automating Transformer Design via Reinforced Architecture Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Guotong Xie, Wei Zhu, Xiaoling Wang, Xipeng Qiu, Yuan Ni","submitted_at":"2020-09-04T08:46:22Z","abstract_excerpt":"Though the transformer architectures have shown dominance in many natural language understanding tasks, there are still unsolved issues for the training of transformer models, especially the need for a principled way of warm-up which has shown importance for stable training of a transformer, as well as whether the task at hand prefer to scale the attention product or not. In this paper, we empirically explore automating the design choices in the transformer model, i.e., how to set layer-norm, whether to scale, number of layers, number of heads, activation function, etc, so that one can obtain "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2009.02070","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-09-04T08:46:22Z","cross_cats_sorted":[],"title_canon_sha256":"bd938b9bb0917199a6260f76f8a355764c1ed46e414fbcf95f758bff216f8119","abstract_canon_sha256":"26786b14373aedecb8e57606b29a0b6eb3ddbf1b62983d7692942be990123487"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:44:33.472299Z","signature_b64":"RKaZY+y/AuKbAtvlhFG1z5/OdFo1yg+UrMEU+cSnDC41catBGxVM4k5uofWgsl8hQukA24DLlO3Ucvy1RipUDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d4d33e8948f8459352882cf88cc8f1e92a931d3ad5e3ab072e51f8d9966305d2","last_reissued_at":"2026-07-05T02:44:33.471887Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:44:33.471887Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AutoTrans: Automating Transformer Design via Reinforced Architecture Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Guotong Xie, Wei Zhu, Xiaoling Wang, Xipeng Qiu, Yuan Ni","submitted_at":"2020-09-04T08:46:22Z","abstract_excerpt":"Though the transformer architectures have shown dominance in many natural language understanding tasks, there are still unsolved issues for the training of transformer models, especially the need for a principled way of warm-up which has shown importance for stable training of a transformer, as well as whether the task at hand prefer to scale the attention product or not. In this paper, we empirically explore automating the design choices in the transformer model, i.e., how to set layer-norm, whether to scale, number of layers, number of heads, activation function, etc, so that one can obtain "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2009.02070","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2009.02070/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2009.02070","created_at":"2026-07-05T02:44:33.471946+00:00"},{"alias_kind":"arxiv_version","alias_value":"2009.02070v2","created_at":"2026-07-05T02:44:33.471946+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2009.02070","created_at":"2026-07-05T02:44:33.471946+00:00"},{"alias_kind":"pith_short_12","alias_value":"2TJT5CKI7BCZ","created_at":"2026-07-05T02:44:33.471946+00:00"},{"alias_kind":"pith_short_16","alias_value":"2TJT5CKI7BCZGUUI","created_at":"2026-07-05T02:44:33.471946+00:00"},{"alias_kind":"pith_short_8","alias_value":"2TJT5CKI","created_at":"2026-07-05T02:44:33.471946+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2TJT5CKI7BCZGUUIFT4IZSHR5E","json":"https://pith.science/pith/2TJT5CKI7BCZGUUIFT4IZSHR5E.json","graph_json":"https://pith.science/api/pith-number/2TJT5CKI7BCZGUUIFT4IZSHR5E/graph.json","events_json":"https://pith.science/api/pith-number/2TJT5CKI7BCZGUUIFT4IZSHR5E/events.json","paper":"https://pith.science/paper/2TJT5CKI"},"agent_actions":{"view_html":"https://pith.science/pith/2TJT5CKI7BCZGUUIFT4IZSHR5E","download_json":"https://pith.science/pith/2TJT5CKI7BCZGUUIFT4IZSHR5E.json","view_paper":"https://pith.science/paper/2TJT5CKI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2009.02070&json=true","fetch_graph":"https://pith.science/api/pith-number/2TJT5CKI7BCZGUUIFT4IZSHR5E/graph.json","fetch_events":"https://pith.science/api/pith-number/2TJT5CKI7BCZGUUIFT4IZSHR5E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2TJT5CKI7BCZGUUIFT4IZSHR5E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2TJT5CKI7BCZGUUIFT4IZSHR5E/action/storage_attestation","attest_author":"https://pith.science/pith/2TJT5CKI7BCZGUUIFT4IZSHR5E/action/author_attestation","sign_citation":"https://pith.science/pith/2TJT5CKI7BCZGUUIFT4IZSHR5E/action/citation_signature","submit_replication":"https://pith.science/pith/2TJT5CKI7BCZGUUIFT4IZSHR5E/action/replication_record"}},"created_at":"2026-07-05T02:44:33.471946+00:00","updated_at":"2026-07-05T02:44:33.471946+00:00"}