{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:LWTPULH3SIZJZ2PCLJQGDF5KQU","short_pith_number":"pith:LWTPULH3","schema_version":"1.0","canonical_sha256":"5da6fa2cfb92329ce9e25a606197aa852f40d40d7ddb418d9ee0436c9436c800","source":{"kind":"arxiv","id":"2106.06981","version":2},"attestation_state":"computed","paper":{"title":"Thinking Like Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Eran Yahav, Gail Weiss, Yoav Goldberg","submitted_at":"2021-06-13T13:04:46Z","abstract_excerpt":"What is the computational model behind a Transformer? Where recurrent neural networks have direct parallels in finite state machines, allowing clear discussion and thought around architecture variants or trained models, Transformers have no such familiar parallel. In this paper we aim to change that, proposing a computational model for the transformer-encoder in the form of a programming language. We map the basic components of a transformer-encoder -- attention and feed-forward computation -- into simple primitives, around which we form a programming language: the Restricted Access Sequence P"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.06981","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-13T13:04:46Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"f23f26182c1bf86ab510d1e8de6c2db1f9bf6a7bc988191c1ec93b25296d0028","abstract_canon_sha256":"0e5daca6b5e42e07f176d3e6acff863761e3709e73010f6b0350dc8f4d2af3a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:58:43.092069Z","signature_b64":"LfQ8Pi+bA7aqVWrJ3N2R+SAjmVOHNSR5hTRO4m6WgILLlOQoKoqtMUaFhEBbOPBejHDjOWppmkn1CRG91giRDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5da6fa2cfb92329ce9e25a606197aa852f40d40d7ddb418d9ee0436c9436c800","last_reissued_at":"2026-07-05T02:58:43.091663Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:58:43.091663Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Thinking Like Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Eran Yahav, Gail Weiss, Yoav Goldberg","submitted_at":"2021-06-13T13:04:46Z","abstract_excerpt":"What is the computational model behind a Transformer? Where recurrent neural networks have direct parallels in finite state machines, allowing clear discussion and thought around architecture variants or trained models, Transformers have no such familiar parallel. In this paper we aim to change that, proposing a computational model for the transformer-encoder in the form of a programming language. We map the basic components of a transformer-encoder -- attention and feed-forward computation -- into simple primitives, around which we form a programming language: the Restricted Access Sequence P"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.06981","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.06981/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.06981","created_at":"2026-07-05T02:58:43.091719+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.06981v2","created_at":"2026-07-05T02:58:43.091719+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.06981","created_at":"2026-07-05T02:58:43.091719+00:00"},{"alias_kind":"pith_short_12","alias_value":"LWTPULH3SIZJ","created_at":"2026-07-05T02:58:43.091719+00:00"},{"alias_kind":"pith_short_16","alias_value":"LWTPULH3SIZJZ2PC","created_at":"2026-07-05T02:58:43.091719+00:00"},{"alias_kind":"pith_short_8","alias_value":"LWTPULH3","created_at":"2026-07-05T02:58:43.091719+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24026","citing_title":"Can Language Model Agents be Helpful Circuit Explainers in Mechanistic Interpretability?","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20936","citing_title":"Comparing Transformers and Hybrid Models at the Token Level","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27206","citing_title":"Syntactic Belief Update as the Driver of Garden Path Processing Difficulty","ref_index":195,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":249,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02371","citing_title":"Internalized Reasoning for Long-Context Visual Document Understanding","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LWTPULH3SIZJZ2PCLJQGDF5KQU","json":"https://pith.science/pith/LWTPULH3SIZJZ2PCLJQGDF5KQU.json","graph_json":"https://pith.science/api/pith-number/LWTPULH3SIZJZ2PCLJQGDF5KQU/graph.json","events_json":"https://pith.science/api/pith-number/LWTPULH3SIZJZ2PCLJQGDF5KQU/events.json","paper":"https://pith.science/paper/LWTPULH3"},"agent_actions":{"view_html":"https://pith.science/pith/LWTPULH3SIZJZ2PCLJQGDF5KQU","download_json":"https://pith.science/pith/LWTPULH3SIZJZ2PCLJQGDF5KQU.json","view_paper":"https://pith.science/paper/LWTPULH3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.06981&json=true","fetch_graph":"https://pith.science/api/pith-number/LWTPULH3SIZJZ2PCLJQGDF5KQU/graph.json","fetch_events":"https://pith.science/api/pith-number/LWTPULH3SIZJZ2PCLJQGDF5KQU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LWTPULH3SIZJZ2PCLJQGDF5KQU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LWTPULH3SIZJZ2PCLJQGDF5KQU/action/storage_attestation","attest_author":"https://pith.science/pith/LWTPULH3SIZJZ2PCLJQGDF5KQU/action/author_attestation","sign_citation":"https://pith.science/pith/LWTPULH3SIZJZ2PCLJQGDF5KQU/action/citation_signature","submit_replication":"https://pith.science/pith/LWTPULH3SIZJZ2PCLJQGDF5KQU/action/replication_record"}},"created_at":"2026-07-05T02:58:43.091719+00:00","updated_at":"2026-07-05T02:58:43.091719+00:00"}