{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MB4NWU57MECKPSCTUIEP2ROFJC","short_pith_number":"pith:MB4NWU57","schema_version":"1.0","canonical_sha256":"6078db53bf6104a7c853a208fd45c548aa253786d25c7f9c6a2a456800c36fe0","source":{"kind":"arxiv","id":"2311.11829","version":1},"attestation_state":"computed","paper":{"title":"System 2 Attention (is something you might need too)","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jason Weston, Sainbayar Sukhbaatar","submitted_at":"2023-11-20T15:04:50Z","abstract_excerpt":"Soft attention in Transformer-based Large Language Models (LLMs) is susceptible to incorporating irrelevant information from the context into its latent representations, which adversely affects next token generations. To help rectify these issues, we introduce System 2 Attention (S2A), which leverages the ability of LLMs to reason in natural language and follow instructions in order to decide what to attend to. S2A regenerates the input context to only include the relevant portions, before attending to the regenerated context to elicit the final response. In experiments, S2A outperforms standa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.11829","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-20T15:04:50Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"2f9bd3c1774125a7999c7fb3ebeae6be5330bc7ed20e75fb688e0bdc79978d67","abstract_canon_sha256":"e77813a81ce21cb713879821a616bf9b35cc54b448b351c671bb75fed62e9755"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:14:39.548737Z","signature_b64":"wUdgedjlYW5UF1lsDZxLwAgfjsB792k1T8HrYakHBB8ZnGoFskiA62pccNEBIPHHEqOfguP60E2cYuEc6DjwAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6078db53bf6104a7c853a208fd45c548aa253786d25c7f9c6a2a456800c36fe0","last_reissued_at":"2026-07-05T07:14:39.548277Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:14:39.548277Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"System 2 Attention (is something you might need too)","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jason Weston, Sainbayar Sukhbaatar","submitted_at":"2023-11-20T15:04:50Z","abstract_excerpt":"Soft attention in Transformer-based Large Language Models (LLMs) is susceptible to incorporating irrelevant information from the context into its latent representations, which adversely affects next token generations. To help rectify these issues, we introduce System 2 Attention (S2A), which leverages the ability of LLMs to reason in natural language and follow instructions in order to decide what to attend to. S2A regenerates the input context to only include the relevant portions, before attending to the regenerated context to elicit the final response. In experiments, S2A outperforms standa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.11829","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.11829/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.11829","created_at":"2026-07-05T07:14:39.548337+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.11829v1","created_at":"2026-07-05T07:14:39.548337+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.11829","created_at":"2026-07-05T07:14:39.548337+00:00"},{"alias_kind":"pith_short_12","alias_value":"MB4NWU57MECK","created_at":"2026-07-05T07:14:39.548337+00:00"},{"alias_kind":"pith_short_16","alias_value":"MB4NWU57MECKPSCT","created_at":"2026-07-05T07:14:39.548337+00:00"},{"alias_kind":"pith_short_8","alias_value":"MB4NWU57","created_at":"2026-07-05T07:14:39.548337+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08018","citing_title":"Concretized Proposition Prompting Resolves Composition-Knowledge Dichotomy in Large Language Models","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2605.23981","citing_title":"Metacognition Should Be the Scientific Framework for Bounded and Effective Self-Governance in Generative AI","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01080","citing_title":"ThinkSwitch: Context Distillation with LoRA and Weight Interpolation for Specific-Purpose Reasoning Tasks","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2402.07927","citing_title":"A Systematic Survey of Prompt Engineering in Large Language Models: Techniques and Applications","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14828","citing_title":"Pangu-ACE: Adaptive Cascaded Experts for Educational Response Generation on EduBench","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MB4NWU57MECKPSCTUIEP2ROFJC","json":"https://pith.science/pith/MB4NWU57MECKPSCTUIEP2ROFJC.json","graph_json":"https://pith.science/api/pith-number/MB4NWU57MECKPSCTUIEP2ROFJC/graph.json","events_json":"https://pith.science/api/pith-number/MB4NWU57MECKPSCTUIEP2ROFJC/events.json","paper":"https://pith.science/paper/MB4NWU57"},"agent_actions":{"view_html":"https://pith.science/pith/MB4NWU57MECKPSCTUIEP2ROFJC","download_json":"https://pith.science/pith/MB4NWU57MECKPSCTUIEP2ROFJC.json","view_paper":"https://pith.science/paper/MB4NWU57","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.11829&json=true","fetch_graph":"https://pith.science/api/pith-number/MB4NWU57MECKPSCTUIEP2ROFJC/graph.json","fetch_events":"https://pith.science/api/pith-number/MB4NWU57MECKPSCTUIEP2ROFJC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MB4NWU57MECKPSCTUIEP2ROFJC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MB4NWU57MECKPSCTUIEP2ROFJC/action/storage_attestation","attest_author":"https://pith.science/pith/MB4NWU57MECKPSCTUIEP2ROFJC/action/author_attestation","sign_citation":"https://pith.science/pith/MB4NWU57MECKPSCTUIEP2ROFJC/action/citation_signature","submit_replication":"https://pith.science/pith/MB4NWU57MECKPSCTUIEP2ROFJC/action/replication_record"}},"created_at":"2026-07-05T07:14:39.548337+00:00","updated_at":"2026-07-05T07:14:39.548337+00:00"}