{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:BX6Z2CHB6B7EQW6SRMSTM644LT","short_pith_number":"pith:BX6Z2CHB","schema_version":"1.0","canonical_sha256":"0dfd9d08e1f07e485bd28b25367b9c5cdec7334b2a70ecde882be8ed6409bc16","source":{"kind":"arxiv","id":"2211.12316","version":2},"attestation_state":"computed","paper":{"title":"Simplicity Bias in Transformers and their Ability to Learn Sparse Boolean Functions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Arkil Patel, Phil Blunsom, Satwik Bhattamishra, Varun Kanade","submitted_at":"2022-11-22T15:10:48Z","abstract_excerpt":"Despite the widespread success of Transformers on NLP tasks, recent works have found that they struggle to model several formal languages when compared to recurrent models. This raises the question of why Transformers perform well in practice and whether they have any properties that enable them to generalize better than recurrent models. In this work, we conduct an extensive empirical study on Boolean functions to demonstrate the following: (i) Random Transformers are relatively more biased towards functions of low sensitivity. (ii) When trained on Boolean functions, both Transformers and LST"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.12316","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-11-22T15:10:48Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"2171fb56ac15fc9c07bc8b2da988b56dfbc9c00f879ca09b61ba6538426c6cef","abstract_canon_sha256":"05852eda7f0e2c67ff5042ebe9c6cb93d3969559296cae99b2efce515542c30a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:28:57.153935Z","signature_b64":"//BLmNWgtAcqWuLiWvIYo7vPrpK/GpHmoF8mC0E7QJzVVHWP3ZnlD8anbaTlv8zPVe/zJkeVUEvc7whIRQt3BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0dfd9d08e1f07e485bd28b25367b9c5cdec7334b2a70ecde882be8ed6409bc16","last_reissued_at":"2026-07-05T06:28:57.153406Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:28:57.153406Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Simplicity Bias in Transformers and their Ability to Learn Sparse Boolean Functions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Arkil Patel, Phil Blunsom, Satwik Bhattamishra, Varun Kanade","submitted_at":"2022-11-22T15:10:48Z","abstract_excerpt":"Despite the widespread success of Transformers on NLP tasks, recent works have found that they struggle to model several formal languages when compared to recurrent models. This raises the question of why Transformers perform well in practice and whether they have any properties that enable them to generalize better than recurrent models. In this work, we conduct an extensive empirical study on Boolean functions to demonstrate the following: (i) Random Transformers are relatively more biased towards functions of low sensitivity. (ii) When trained on Boolean functions, both Transformers and LST"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.12316","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.12316/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.12316","created_at":"2026-07-05T06:28:57.153467+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.12316v2","created_at":"2026-07-05T06:28:57.153467+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.12316","created_at":"2026-07-05T06:28:57.153467+00:00"},{"alias_kind":"pith_short_12","alias_value":"BX6Z2CHB6B7E","created_at":"2026-07-05T06:28:57.153467+00:00"},{"alias_kind":"pith_short_16","alias_value":"BX6Z2CHB6B7EQW6S","created_at":"2026-07-05T06:28:57.153467+00:00"},{"alias_kind":"pith_short_8","alias_value":"BX6Z2CHB","created_at":"2026-07-05T06:28:57.153467+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00605","citing_title":"Looped Transformers with Layer Normalization Provably Learn the Power Method","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2601.03162","citing_title":"On the Convergence Behavior of Preconditioned Gradient Descent Toward the Rich Learning Regime","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10019","citing_title":"The two clocks and the innovation window: When and how generative models learn rules","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BX6Z2CHB6B7EQW6SRMSTM644LT","json":"https://pith.science/pith/BX6Z2CHB6B7EQW6SRMSTM644LT.json","graph_json":"https://pith.science/api/pith-number/BX6Z2CHB6B7EQW6SRMSTM644LT/graph.json","events_json":"https://pith.science/api/pith-number/BX6Z2CHB6B7EQW6SRMSTM644LT/events.json","paper":"https://pith.science/paper/BX6Z2CHB"},"agent_actions":{"view_html":"https://pith.science/pith/BX6Z2CHB6B7EQW6SRMSTM644LT","download_json":"https://pith.science/pith/BX6Z2CHB6B7EQW6SRMSTM644LT.json","view_paper":"https://pith.science/paper/BX6Z2CHB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.12316&json=true","fetch_graph":"https://pith.science/api/pith-number/BX6Z2CHB6B7EQW6SRMSTM644LT/graph.json","fetch_events":"https://pith.science/api/pith-number/BX6Z2CHB6B7EQW6SRMSTM644LT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BX6Z2CHB6B7EQW6SRMSTM644LT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BX6Z2CHB6B7EQW6SRMSTM644LT/action/storage_attestation","attest_author":"https://pith.science/pith/BX6Z2CHB6B7EQW6SRMSTM644LT/action/author_attestation","sign_citation":"https://pith.science/pith/BX6Z2CHB6B7EQW6SRMSTM644LT/action/citation_signature","submit_replication":"https://pith.science/pith/BX6Z2CHB6B7EQW6SRMSTM644LT/action/replication_record"}},"created_at":"2026-07-05T06:28:57.153467+00:00","updated_at":"2026-07-05T06:28:57.153467+00:00"}