{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2SB3T4LP6UHPLPUUJKHNM3KSNA","short_pith_number":"pith:2SB3T4LP","schema_version":"1.0","canonical_sha256":"d483b9f16ff50ef5be944a8ed66d526824216d81b22902b41e8ca4cf0166e1aa","source":{"kind":"arxiv","id":"2410.10986","version":2},"attestation_state":"computed","paper":{"title":"What Does It Mean to Be a Transformer? Insights from a Theoretical Hessian Analysis","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Felix Dangel, Sidak Pal Singh, Weronika Ormaniec","submitted_at":"2024-10-14T18:15:02Z","abstract_excerpt":"The Transformer architecture has inarguably revolutionized deep learning, overtaking classical architectures like multi-layer perceptrons (MLPs) and convolutional neural networks (CNNs). At its core, the attention block differs in form and functionality from most other architectural components in deep learning--to the extent that, in comparison to MLPs/CNNs, Transformers are more often accompanied by adaptive optimizers, layer normalization, learning rate warmup, etc. The root causes behind these outward manifestations and the precise mechanisms that govern them remain poorly understood. In th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.10986","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-14T18:15:02Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"f1aaf5ece8743fc2526d73885a41e9f73ebf1debe4cbdc40e611e71c9888d4a4","abstract_canon_sha256":"8eb50b46ef2fc38f53b727cf92bbedb83f8ed96dc4e94590d489ff3e3aafcf8a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:32:32.136371Z","signature_b64":"K+YQxI0Fp3zaMrBfqCYy/qeUY1xFVcyzHuos5ysahms+dWoG1G1LnkU1YnZWm2c9HSDNB4UM8ndzeQ6e2K2uAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d483b9f16ff50ef5be944a8ed66d526824216d81b22902b41e8ca4cf0166e1aa","last_reissued_at":"2026-07-05T10:32:32.135843Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:32:32.135843Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Does It Mean to Be a Transformer? Insights from a Theoretical Hessian Analysis","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Felix Dangel, Sidak Pal Singh, Weronika Ormaniec","submitted_at":"2024-10-14T18:15:02Z","abstract_excerpt":"The Transformer architecture has inarguably revolutionized deep learning, overtaking classical architectures like multi-layer perceptrons (MLPs) and convolutional neural networks (CNNs). At its core, the attention block differs in form and functionality from most other architectural components in deep learning--to the extent that, in comparison to MLPs/CNNs, Transformers are more often accompanied by adaptive optimizers, layer normalization, learning rate warmup, etc. The root causes behind these outward manifestations and the precise mechanisms that govern them remain poorly understood. In th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.10986","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.10986/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.10986","created_at":"2026-07-05T10:32:32.135906+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.10986v2","created_at":"2026-07-05T10:32:32.135906+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.10986","created_at":"2026-07-05T10:32:32.135906+00:00"},{"alias_kind":"pith_short_12","alias_value":"2SB3T4LP6UHP","created_at":"2026-07-05T10:32:32.135906+00:00"},{"alias_kind":"pith_short_16","alias_value":"2SB3T4LP6UHPLPUU","created_at":"2026-07-05T10:32:32.135906+00:00"},{"alias_kind":"pith_short_8","alias_value":"2SB3T4LP","created_at":"2026-07-05T10:32:32.135906+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.05794","citing_title":"Revealing Modular Gradient Noise Imbalance in LLMs: Calibrating Adam via Signal-to-Noise Ratio","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07959","citing_title":"Convergent Stochastic Training of Attention and Understanding LoRA","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2SB3T4LP6UHPLPUUJKHNM3KSNA","json":"https://pith.science/pith/2SB3T4LP6UHPLPUUJKHNM3KSNA.json","graph_json":"https://pith.science/api/pith-number/2SB3T4LP6UHPLPUUJKHNM3KSNA/graph.json","events_json":"https://pith.science/api/pith-number/2SB3T4LP6UHPLPUUJKHNM3KSNA/events.json","paper":"https://pith.science/paper/2SB3T4LP"},"agent_actions":{"view_html":"https://pith.science/pith/2SB3T4LP6UHPLPUUJKHNM3KSNA","download_json":"https://pith.science/pith/2SB3T4LP6UHPLPUUJKHNM3KSNA.json","view_paper":"https://pith.science/paper/2SB3T4LP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.10986&json=true","fetch_graph":"https://pith.science/api/pith-number/2SB3T4LP6UHPLPUUJKHNM3KSNA/graph.json","fetch_events":"https://pith.science/api/pith-number/2SB3T4LP6UHPLPUUJKHNM3KSNA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2SB3T4LP6UHPLPUUJKHNM3KSNA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2SB3T4LP6UHPLPUUJKHNM3KSNA/action/storage_attestation","attest_author":"https://pith.science/pith/2SB3T4LP6UHPLPUUJKHNM3KSNA/action/author_attestation","sign_citation":"https://pith.science/pith/2SB3T4LP6UHPLPUUJKHNM3KSNA/action/citation_signature","submit_replication":"https://pith.science/pith/2SB3T4LP6UHPLPUUJKHNM3KSNA/action/replication_record"}},"created_at":"2026-07-05T10:32:32.135906+00:00","updated_at":"2026-07-05T10:32:32.135906+00:00"}