{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IYOOFC236ML6QGUYI6XRH33YUF","short_pith_number":"pith:IYOOFC23","schema_version":"1.0","canonical_sha256":"461ce28b5bf317e81a9847af13ef78a147d58d33428c4d6542f279b087caeaab","source":{"kind":"arxiv","id":"2509.09650","version":1},"attestation_state":"computed","paper":{"title":"All for One: LLMs Solve Mental Math at the Last Token With Information Transferred From Other Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Daking Rai, Siddarth Mamidanna, Yilun Zhou, Ziyu Yao","submitted_at":"2025-09-11T17:41:29Z","abstract_excerpt":"Large language models (LLMs) demonstrate proficiency across numerous computational tasks, yet their inner workings remain unclear. In theory, the combination of causal self-attention and multilayer perceptron layers allows every token to access and compute information based on all preceding tokens. In practice, to what extent are such operations present? In this paper, on mental math tasks (i.e., direct math calculation via next-token prediction without explicit reasoning), we investigate this question in three steps: inhibiting input-specific token computations in the initial layers, restrict"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.09650","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-09-11T17:41:29Z","cross_cats_sorted":[],"title_canon_sha256":"6074f92f1bf8e8fc6bdea85673cbfe2d78ab69149183c9a64c6b75234f9110d2","abstract_canon_sha256":"e702aeb1f1e0a2d1c3629b2806e7a0f1998681f97844decb323355f2adb6965e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:09:40.255975Z","signature_b64":"FlmKsiinmPb278cQ0xtVhDY48OI/UtIG5VgoSr5hJlrtwg009Sd7Z5lI9sVaTvPZ7HXoch7blVVm9Fx5CBenCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"461ce28b5bf317e81a9847af13ef78a147d58d33428c4d6542f279b087caeaab","last_reissued_at":"2026-07-05T12:09:40.255472Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:09:40.255472Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"All for One: LLMs Solve Mental Math at the Last Token With Information Transferred From Other Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Daking Rai, Siddarth Mamidanna, Yilun Zhou, Ziyu Yao","submitted_at":"2025-09-11T17:41:29Z","abstract_excerpt":"Large language models (LLMs) demonstrate proficiency across numerous computational tasks, yet their inner workings remain unclear. In theory, the combination of causal self-attention and multilayer perceptron layers allows every token to access and compute information based on all preceding tokens. In practice, to what extent are such operations present? In this paper, on mental math tasks (i.e., direct math calculation via next-token prediction without explicit reasoning), we investigate this question in three steps: inhibiting input-specific token computations in the initial layers, restrict"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.09650","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.09650/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.09650","created_at":"2026-07-05T12:09:40.255533+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.09650v1","created_at":"2026-07-05T12:09:40.255533+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.09650","created_at":"2026-07-05T12:09:40.255533+00:00"},{"alias_kind":"pith_short_12","alias_value":"IYOOFC236ML6","created_at":"2026-07-05T12:09:40.255533+00:00"},{"alias_kind":"pith_short_16","alias_value":"IYOOFC236ML6QGUY","created_at":"2026-07-05T12:09:40.255533+00:00"},{"alias_kind":"pith_short_8","alias_value":"IYOOFC23","created_at":"2026-07-05T12:09:40.255533+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.14685","citing_title":"SSA: Improving Performance With a Better Scoring Function","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IYOOFC236ML6QGUYI6XRH33YUF","json":"https://pith.science/pith/IYOOFC236ML6QGUYI6XRH33YUF.json","graph_json":"https://pith.science/api/pith-number/IYOOFC236ML6QGUYI6XRH33YUF/graph.json","events_json":"https://pith.science/api/pith-number/IYOOFC236ML6QGUYI6XRH33YUF/events.json","paper":"https://pith.science/paper/IYOOFC23"},"agent_actions":{"view_html":"https://pith.science/pith/IYOOFC236ML6QGUYI6XRH33YUF","download_json":"https://pith.science/pith/IYOOFC236ML6QGUYI6XRH33YUF.json","view_paper":"https://pith.science/paper/IYOOFC23","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.09650&json=true","fetch_graph":"https://pith.science/api/pith-number/IYOOFC236ML6QGUYI6XRH33YUF/graph.json","fetch_events":"https://pith.science/api/pith-number/IYOOFC236ML6QGUYI6XRH33YUF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IYOOFC236ML6QGUYI6XRH33YUF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IYOOFC236ML6QGUYI6XRH33YUF/action/storage_attestation","attest_author":"https://pith.science/pith/IYOOFC236ML6QGUYI6XRH33YUF/action/author_attestation","sign_citation":"https://pith.science/pith/IYOOFC236ML6QGUYI6XRH33YUF/action/citation_signature","submit_replication":"https://pith.science/pith/IYOOFC236ML6QGUYI6XRH33YUF/action/replication_record"}},"created_at":"2026-07-05T12:09:40.255533+00:00","updated_at":"2026-07-05T12:09:40.255533+00:00"}