{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DBJMWF7CWHZNOWVVWQTU4HEBHV","short_pith_number":"pith:DBJMWF7C","schema_version":"1.0","canonical_sha256":"1852cb17e2b1f2d75ab5b4274e1c813d48773fa9c0b532a5f90a69f8138d8a31","source":{"kind":"arxiv","id":"2402.09963","version":4},"attestation_state":"computed","paper":{"title":"Why are Sensitive Functions Hard for Transformers?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Mark Rofin, Michael Hahn","submitted_at":"2024-02-15T14:17:51Z","abstract_excerpt":"Empirical studies have identified a range of learnability biases and limitations of transformers, such as a persistent difficulty in learning to compute simple formal languages such as PARITY, and a bias towards low-degree functions. However, theoretical understanding remains limited, with existing expressiveness theory either overpredicting or underpredicting realistic learning abilities. We prove that, under the transformer architecture, the loss landscape is constrained by the input-space sensitivity: Transformers whose output is sensitive to many parts of the input string inhabit isolated "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.09963","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-15T14:17:51Z","cross_cats_sorted":[],"title_canon_sha256":"40fe2b564024af886a4b2cb385ecda164be4c5e7bd4975c29060692418e706f0","abstract_canon_sha256":"4f245183dfc449d721f768026df200ecc7cdc89f814ef070489bc8d42395fccd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:18.099532Z","signature_b64":"KKsul8pSVOrtTT4z7L5YtZCfczbx/sA1DNBKXDojpiWa/NhF6vzgzDzyXqtp7rJs/cjooiFnRdL9azbKVXAyAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1852cb17e2b1f2d75ab5b4274e1c813d48773fa9c0b532a5f90a69f8138d8a31","last_reissued_at":"2026-07-05T08:23:18.099049Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:18.099049Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Why are Sensitive Functions Hard for Transformers?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Mark Rofin, Michael Hahn","submitted_at":"2024-02-15T14:17:51Z","abstract_excerpt":"Empirical studies have identified a range of learnability biases and limitations of transformers, such as a persistent difficulty in learning to compute simple formal languages such as PARITY, and a bias towards low-degree functions. However, theoretical understanding remains limited, with existing expressiveness theory either overpredicting or underpredicting realistic learning abilities. We prove that, under the transformer architecture, the loss landscape is constrained by the input-space sensitivity: Transformers whose output is sensitive to many parts of the input string inhabit isolated "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.09963","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.09963/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.09963","created_at":"2026-07-05T08:23:18.099104+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.09963v4","created_at":"2026-07-05T08:23:18.099104+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.09963","created_at":"2026-07-05T08:23:18.099104+00:00"},{"alias_kind":"pith_short_12","alias_value":"DBJMWF7CWHZN","created_at":"2026-07-05T08:23:18.099104+00:00"},{"alias_kind":"pith_short_16","alias_value":"DBJMWF7CWHZNOWVV","created_at":"2026-07-05T08:23:18.099104+00:00"},{"alias_kind":"pith_short_8","alias_value":"DBJMWF7C","created_at":"2026-07-05T08:23:18.099104+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00183","citing_title":"Agentic Transformers Provably Learn to Search via Reinforcement Learning","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01651","citing_title":"On the Spatiotemporal Dynamics of Generalization in Neural Networks","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2603.29069","citing_title":"On the Mirage of Long-Range Dependency, with an Application to Integer Multiplication","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10237","citing_title":"The Benefits of Temporal Correlations: SGD Learns k-Juntas from Random Walks Efficiently","ref_index":128,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01381","citing_title":"A framework for analyzing concept representations in neural models","ref_index":242,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DBJMWF7CWHZNOWVVWQTU4HEBHV","json":"https://pith.science/pith/DBJMWF7CWHZNOWVVWQTU4HEBHV.json","graph_json":"https://pith.science/api/pith-number/DBJMWF7CWHZNOWVVWQTU4HEBHV/graph.json","events_json":"https://pith.science/api/pith-number/DBJMWF7CWHZNOWVVWQTU4HEBHV/events.json","paper":"https://pith.science/paper/DBJMWF7C"},"agent_actions":{"view_html":"https://pith.science/pith/DBJMWF7CWHZNOWVVWQTU4HEBHV","download_json":"https://pith.science/pith/DBJMWF7CWHZNOWVVWQTU4HEBHV.json","view_paper":"https://pith.science/paper/DBJMWF7C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.09963&json=true","fetch_graph":"https://pith.science/api/pith-number/DBJMWF7CWHZNOWVVWQTU4HEBHV/graph.json","fetch_events":"https://pith.science/api/pith-number/DBJMWF7CWHZNOWVVWQTU4HEBHV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DBJMWF7CWHZNOWVVWQTU4HEBHV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DBJMWF7CWHZNOWVVWQTU4HEBHV/action/storage_attestation","attest_author":"https://pith.science/pith/DBJMWF7CWHZNOWVVWQTU4HEBHV/action/author_attestation","sign_citation":"https://pith.science/pith/DBJMWF7CWHZNOWVVWQTU4HEBHV/action/citation_signature","submit_replication":"https://pith.science/pith/DBJMWF7CWHZNOWVVWQTU4HEBHV/action/replication_record"}},"created_at":"2026-07-05T08:23:18.099104+00:00","updated_at":"2026-07-05T08:23:18.099104+00:00"}