{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ZTIQGL6GKEYPB7MWKBQSXFIN4K","short_pith_number":"pith:ZTIQGL6G","schema_version":"1.0","canonical_sha256":"ccd1032fc65130f0fd9650612b950de2a7449d1f8a25e775171cc2af3729270e","source":{"kind":"arxiv","id":"2206.04301","version":3},"attestation_state":"computed","paper":{"title":"Unveiling Transformers with LEGO: a synthetic reasoning task","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Arturs Backurs, Ronen Eldan, S\\'ebastien Bubeck, Suriya Gunasekar, Tal Wagner, Yi Zhang","submitted_at":"2022-06-09T06:30:17Z","abstract_excerpt":"We propose a synthetic reasoning task, LEGO (Learning Equality and Group Operations), that encapsulates the problem of following a chain of reasoning, and we study how the Transformer architectures learn this task. We pay special attention to data effects such as pretraining (on seemingly unrelated NLP tasks) and dataset composition (e.g., differing chain length at training and test time), as well as architectural variants such as weight-tied layers or adding convolutional components. We study how the trained models eventually succeed at the task, and in particular, we manage to understand som"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.04301","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-09T06:30:17Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"0645e53101edfc5e56238b2b22919fe33c5e1bb15fd8b44c9ce13038fe68927f","abstract_canon_sha256":"251e1e5339e8c4148b3a50b15300830dce629e213a14f77f7c50a9492e9a90f1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:43:11.823501Z","signature_b64":"P9VBUxlTIjphtNu4weuEyPrdiX7li2rdCVVm7XJQvJgyqor8OUlj45PC2C5SvWTzf+pMwN4KijmYkcY1RXKXDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ccd1032fc65130f0fd9650612b950de2a7449d1f8a25e775171cc2af3729270e","last_reissued_at":"2026-07-05T05:43:11.823053Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:43:11.823053Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unveiling Transformers with LEGO: a synthetic reasoning task","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Arturs Backurs, Ronen Eldan, S\\'ebastien Bubeck, Suriya Gunasekar, Tal Wagner, Yi Zhang","submitted_at":"2022-06-09T06:30:17Z","abstract_excerpt":"We propose a synthetic reasoning task, LEGO (Learning Equality and Group Operations), that encapsulates the problem of following a chain of reasoning, and we study how the Transformer architectures learn this task. We pay special attention to data effects such as pretraining (on seemingly unrelated NLP tasks) and dataset composition (e.g., differing chain length at training and test time), as well as architectural variants such as weight-tied layers or adding convolutional components. We study how the trained models eventually succeed at the task, and in particular, we manage to understand som"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.04301","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.04301/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.04301","created_at":"2026-07-05T05:43:11.823108+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.04301v3","created_at":"2026-07-05T05:43:11.823108+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.04301","created_at":"2026-07-05T05:43:11.823108+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZTIQGL6GKEYP","created_at":"2026-07-05T05:43:11.823108+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZTIQGL6GKEYPB7MW","created_at":"2026-07-05T05:43:11.823108+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZTIQGL6G","created_at":"2026-07-05T05:43:11.823108+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28560","citing_title":"Depth-Staggered Fibonacci Spacing for Sparse Attention: Static Schedules Beat Learned Dilation and Extrapolate Where Dense Attention Fails","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31497","citing_title":"Assign and Add: A Mechanistic Study of Compositional Arithmetic","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2402.17762","citing_title":"Massive Activations in Large Language Models","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2211.00593","citing_title":"Interpretability in the Wild: a Circuit for Indirect Object Identification in GPT-2 small","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05495","citing_title":"Shortcut Solutions Learned by Transformers Impair Continual Compositional Reasoning","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZTIQGL6GKEYPB7MWKBQSXFIN4K","json":"https://pith.science/pith/ZTIQGL6GKEYPB7MWKBQSXFIN4K.json","graph_json":"https://pith.science/api/pith-number/ZTIQGL6GKEYPB7MWKBQSXFIN4K/graph.json","events_json":"https://pith.science/api/pith-number/ZTIQGL6GKEYPB7MWKBQSXFIN4K/events.json","paper":"https://pith.science/paper/ZTIQGL6G"},"agent_actions":{"view_html":"https://pith.science/pith/ZTIQGL6GKEYPB7MWKBQSXFIN4K","download_json":"https://pith.science/pith/ZTIQGL6GKEYPB7MWKBQSXFIN4K.json","view_paper":"https://pith.science/paper/ZTIQGL6G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.04301&json=true","fetch_graph":"https://pith.science/api/pith-number/ZTIQGL6GKEYPB7MWKBQSXFIN4K/graph.json","fetch_events":"https://pith.science/api/pith-number/ZTIQGL6GKEYPB7MWKBQSXFIN4K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZTIQGL6GKEYPB7MWKBQSXFIN4K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZTIQGL6GKEYPB7MWKBQSXFIN4K/action/storage_attestation","attest_author":"https://pith.science/pith/ZTIQGL6GKEYPB7MWKBQSXFIN4K/action/author_attestation","sign_citation":"https://pith.science/pith/ZTIQGL6GKEYPB7MWKBQSXFIN4K/action/citation_signature","submit_replication":"https://pith.science/pith/ZTIQGL6GKEYPB7MWKBQSXFIN4K/action/replication_record"}},"created_at":"2026-07-05T05:43:11.823108+00:00","updated_at":"2026-07-05T05:43:11.823108+00:00"}