{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YQJY37MYHTMY2OS45UCRD4FQLL","short_pith_number":"pith:YQJY37MY","schema_version":"1.0","canonical_sha256":"c4138dfd983cd98d3a5ced0511f0b05af06a9c84106d49a6bc75c5b881acbc4d","source":{"kind":"arxiv","id":"2310.13121","version":9},"attestation_state":"computed","paper":{"title":"Understanding Addition in Transformers","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Fazl Barez, Philip Quirke","submitted_at":"2023-10-19T19:34:42Z","abstract_excerpt":"Understanding the inner workings of machine learning models like Transformers is vital for their safe and ethical use. This paper provides a comprehensive analysis of a one-layer Transformer model trained to perform n-digit integer addition. Our findings suggest that the model dissects the task into parallel streams dedicated to individual digits, employing varied algorithms tailored to different positions within the digits. Furthermore, we identify a rare scenario characterized by high loss, which we explain. By thoroughly elucidating the model's algorithm, we provide new insights into its fu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.13121","kind":"arxiv","version":9},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-19T19:34:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"50fc6c2ac661fa5b20b7f1d7dd679ef76aad745d676f64c42a0c199aabeb37da","abstract_canon_sha256":"c8b31c560f42dcb1334c81c489b4a874b432f705b8b85b572913e370b6e65c96"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:11:20.918492Z","signature_b64":"ZsXpw8tjMLr2WUrhXPyDpn3ihA1bns3+ntm2K4U20ZjkEfbdPAyNo5qu1uKvdIvVwaJIvDOmKoxNylADxaTjAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c4138dfd983cd98d3a5ced0511f0b05af06a9c84106d49a6bc75c5b881acbc4d","last_reissued_at":"2026-07-05T08:11:20.917972Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:11:20.917972Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Addition in Transformers","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Fazl Barez, Philip Quirke","submitted_at":"2023-10-19T19:34:42Z","abstract_excerpt":"Understanding the inner workings of machine learning models like Transformers is vital for their safe and ethical use. This paper provides a comprehensive analysis of a one-layer Transformer model trained to perform n-digit integer addition. Our findings suggest that the model dissects the task into parallel streams dedicated to individual digits, employing varied algorithms tailored to different positions within the digits. Furthermore, we identify a rare scenario characterized by high loss, which we explain. By thoroughly elucidating the model's algorithm, we provide new insights into its fu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.13121","kind":"arxiv","version":9},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.13121/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.13121","created_at":"2026-07-05T08:11:20.918032+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.13121v9","created_at":"2026-07-05T08:11:20.918032+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.13121","created_at":"2026-07-05T08:11:20.918032+00:00"},{"alias_kind":"pith_short_12","alias_value":"YQJY37MYHTMY","created_at":"2026-07-05T08:11:20.918032+00:00"},{"alias_kind":"pith_short_16","alias_value":"YQJY37MYHTMY2OS4","created_at":"2026-07-05T08:11:20.918032+00:00"},{"alias_kind":"pith_short_8","alias_value":"YQJY37MY","created_at":"2026-07-05T08:11:20.918032+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.13082","citing_title":"The Long Delay to Arithmetic Generalization: When Learned Representations Outrun Behavior","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15306","citing_title":"Generalization in LLM Problem Solving: The Case of the Shortest Path","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YQJY37MYHTMY2OS45UCRD4FQLL","json":"https://pith.science/pith/YQJY37MYHTMY2OS45UCRD4FQLL.json","graph_json":"https://pith.science/api/pith-number/YQJY37MYHTMY2OS45UCRD4FQLL/graph.json","events_json":"https://pith.science/api/pith-number/YQJY37MYHTMY2OS45UCRD4FQLL/events.json","paper":"https://pith.science/paper/YQJY37MY"},"agent_actions":{"view_html":"https://pith.science/pith/YQJY37MYHTMY2OS45UCRD4FQLL","download_json":"https://pith.science/pith/YQJY37MYHTMY2OS45UCRD4FQLL.json","view_paper":"https://pith.science/paper/YQJY37MY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.13121&json=true","fetch_graph":"https://pith.science/api/pith-number/YQJY37MYHTMY2OS45UCRD4FQLL/graph.json","fetch_events":"https://pith.science/api/pith-number/YQJY37MYHTMY2OS45UCRD4FQLL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YQJY37MYHTMY2OS45UCRD4FQLL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YQJY37MYHTMY2OS45UCRD4FQLL/action/storage_attestation","attest_author":"https://pith.science/pith/YQJY37MYHTMY2OS45UCRD4FQLL/action/author_attestation","sign_citation":"https://pith.science/pith/YQJY37MYHTMY2OS45UCRD4FQLL/action/citation_signature","submit_replication":"https://pith.science/pith/YQJY37MYHTMY2OS45UCRD4FQLL/action/replication_record"}},"created_at":"2026-07-05T08:11:20.918032+00:00","updated_at":"2026-07-05T08:11:20.918032+00:00"}