{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IPC4VQRJSO46JEBECMR4YVKWWB","short_pith_number":"pith:IPC4VQRJ","schema_version":"1.0","canonical_sha256":"43c5cac22993b9e490241323cc5556b04b3cae2287c46236087cb3f9ce75079a","source":{"kind":"arxiv","id":"2506.01115","version":3},"attestation_state":"computed","paper":{"title":"Is Random Attention Sufficient for Sequence Modeling? Disentangling Trainable Components in the Transformer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Lorenzo Noci, Mikhail Khodak, Mufan Li, Yihe Dong","submitted_at":"2025-06-01T18:42:39Z","abstract_excerpt":"The transformer architecture is central to the success of modern Large Language Models (LLMs), in part due to its surprising ability to perform a wide range of tasks - including mathematical reasoning, memorization, and retrieval - using only gradient-based learning on next-token prediction. While the core component of a transformer is the self-attention mechanism, we question how much, and which aspects, of the performance gains can be attributed to it. To this end, we compare standard transformers to variants in which either the MLP layers or the attention weights are frozen at initializatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.01115","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-01T18:42:39Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"ea3b1c7e2906712608ecfe1e3a65513437f37f624bbd786a737b7b4d39dae8e2","abstract_canon_sha256":"74e342af4ecb86ca212c685e809068a6366a43840e57f8454fdb64339f397383"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:04:35.385635Z","signature_b64":"fMW2pzwsfml2izF9W/o44/49OKZ0pPfH3RBiXyjLQquJImf/obuwRTdLW/WyucQq8jjjKVoZG6sIGorGXXxhCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43c5cac22993b9e490241323cc5556b04b3cae2287c46236087cb3f9ce75079a","last_reissued_at":"2026-07-05T12:04:35.385022Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:04:35.385022Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is Random Attention Sufficient for Sequence Modeling? Disentangling Trainable Components in the Transformer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Lorenzo Noci, Mikhail Khodak, Mufan Li, Yihe Dong","submitted_at":"2025-06-01T18:42:39Z","abstract_excerpt":"The transformer architecture is central to the success of modern Large Language Models (LLMs), in part due to its surprising ability to perform a wide range of tasks - including mathematical reasoning, memorization, and retrieval - using only gradient-based learning on next-token prediction. While the core component of a transformer is the self-attention mechanism, we question how much, and which aspects, of the performance gains can be attributed to it. To this end, we compare standard transformers to variants in which either the MLP layers or the attention weights are frozen at initializatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.01115","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.01115/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.01115","created_at":"2026-07-05T12:04:35.385099+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.01115v3","created_at":"2026-07-05T12:04:35.385099+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.01115","created_at":"2026-07-05T12:04:35.385099+00:00"},{"alias_kind":"pith_short_12","alias_value":"IPC4VQRJSO46","created_at":"2026-07-05T12:04:35.385099+00:00"},{"alias_kind":"pith_short_16","alias_value":"IPC4VQRJSO46JEBE","created_at":"2026-07-05T12:04:35.385099+00:00"},{"alias_kind":"pith_short_8","alias_value":"IPC4VQRJ","created_at":"2026-07-05T12:04:35.385099+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05134","citing_title":"Activation-Based Active Learning for In-Context Learning: Challenges and Insights","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31423","citing_title":"Fixed Universal Transformers","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12744","citing_title":"Resting Neurons, Active Insights: Robustifying Activation Sparsity in LLMs via Spontaneity","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2508.00901","citing_title":"Provable Knowledge Acquisition and Extraction in One-Layer Transformers","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12744","citing_title":"Resting Neurons, Active Insights: Robustifying Activation Sparsity in LLMs via Spontaneity","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05686","citing_title":"Attractor Geometry of Transformer Memory: From Conflict Arbitration to Confident Hallucination","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27914","citing_title":"Geometry-Calibrated Conformal Abstention for Language Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05686","citing_title":"Attractor Geometry of Transformer Memory: From Conflict Arbitration to Confident Hallucination","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IPC4VQRJSO46JEBECMR4YVKWWB","json":"https://pith.science/pith/IPC4VQRJSO46JEBECMR4YVKWWB.json","graph_json":"https://pith.science/api/pith-number/IPC4VQRJSO46JEBECMR4YVKWWB/graph.json","events_json":"https://pith.science/api/pith-number/IPC4VQRJSO46JEBECMR4YVKWWB/events.json","paper":"https://pith.science/paper/IPC4VQRJ"},"agent_actions":{"view_html":"https://pith.science/pith/IPC4VQRJSO46JEBECMR4YVKWWB","download_json":"https://pith.science/pith/IPC4VQRJSO46JEBECMR4YVKWWB.json","view_paper":"https://pith.science/paper/IPC4VQRJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.01115&json=true","fetch_graph":"https://pith.science/api/pith-number/IPC4VQRJSO46JEBECMR4YVKWWB/graph.json","fetch_events":"https://pith.science/api/pith-number/IPC4VQRJSO46JEBECMR4YVKWWB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IPC4VQRJSO46JEBECMR4YVKWWB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IPC4VQRJSO46JEBECMR4YVKWWB/action/storage_attestation","attest_author":"https://pith.science/pith/IPC4VQRJSO46JEBECMR4YVKWWB/action/author_attestation","sign_citation":"https://pith.science/pith/IPC4VQRJSO46JEBECMR4YVKWWB/action/citation_signature","submit_replication":"https://pith.science/pith/IPC4VQRJSO46JEBECMR4YVKWWB/action/replication_record"}},"created_at":"2026-07-05T12:04:35.385099+00:00","updated_at":"2026-07-05T12:04:35.385099+00:00"}