{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VQI3MMGEMTG62Y2CUXB4GT5ULX","short_pith_number":"pith:VQI3MMGE","schema_version":"1.0","canonical_sha256":"ac11b630c464cded6342a5c3c34fb45dce481c3e7c1d7e1e4335fb450235f113","source":{"kind":"arxiv","id":"2309.12288","version":4},"attestation_state":"computed","paper":{"title":"The Reversal Curse: LLMs trained on \"A is B\" fail to learn \"B is A\"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Asa Cooper Stickland, Lukas Berglund, Max Kaufmann, Meg Tong, Mikita Balesni, Owain Evans, Tomasz Korbak","submitted_at":"2023-09-21T17:52:19Z","abstract_excerpt":"We expose a surprising failure of generalization in auto-regressive large language models (LLMs). If a model is trained on a sentence of the form \"A is B\", it will not automatically generalize to the reverse direction \"B is A\". This is the Reversal Curse. For instance, if a model is trained on \"Valentina Tereshkova was the first woman to travel to space\", it will not automatically be able to answer the question, \"Who was the first woman to travel to space?\". Moreover, the likelihood of the correct answer (\"Valentina Tershkova\") will not be higher than for a random name. Thus, models do not gen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.12288","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-21T17:52:19Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"a0c1f319f9b6da222c6162070cf795c35be1b25cc941e9a05930700c0ecac2d9","abstract_canon_sha256":"bd1683d96e7baa1145840449196990e41229da56564788ba2ee53f40d57f87e5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:12.188579Z","signature_b64":"d0v2ZJqC2DEk6IesbVYJPAdblrfQl5WRhR/kMwk7NZaE21BvW3a9B2KbLgjYj2JSkPBMz8U3X2tjhPNVN0geCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac11b630c464cded6342a5c3c34fb45dce481c3e7c1d7e1e4335fb450235f113","last_reissued_at":"2026-07-05T08:23:12.188101Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:12.188101Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Reversal Curse: LLMs trained on \"A is B\" fail to learn \"B is A\"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Asa Cooper Stickland, Lukas Berglund, Max Kaufmann, Meg Tong, Mikita Balesni, Owain Evans, Tomasz Korbak","submitted_at":"2023-09-21T17:52:19Z","abstract_excerpt":"We expose a surprising failure of generalization in auto-regressive large language models (LLMs). If a model is trained on a sentence of the form \"A is B\", it will not automatically generalize to the reverse direction \"B is A\". This is the Reversal Curse. For instance, if a model is trained on \"Valentina Tereshkova was the first woman to travel to space\", it will not automatically be able to answer the question, \"Who was the first woman to travel to space?\". Moreover, the likelihood of the correct answer (\"Valentina Tershkova\") will not be higher than for a random name. Thus, models do not gen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.12288","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.12288/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.12288","created_at":"2026-07-05T08:23:12.188156+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.12288v4","created_at":"2026-07-05T08:23:12.188156+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.12288","created_at":"2026-07-05T08:23:12.188156+00:00"},{"alias_kind":"pith_short_12","alias_value":"VQI3MMGEMTG6","created_at":"2026-07-05T08:23:12.188156+00:00"},{"alias_kind":"pith_short_16","alias_value":"VQI3MMGEMTG62Y2C","created_at":"2026-07-05T08:23:12.188156+00:00"},{"alias_kind":"pith_short_8","alias_value":"VQI3MMGE","created_at":"2026-07-05T08:23:12.188156+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27361","citing_title":"Autoregressive Boltzmann Generators","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19941","citing_title":"Compositionality Emerges in a Narrow Depth-Connectivity Regime: Architecture Constraints and Solution Manifolds","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12504","citing_title":"A Type Theory of Sense: Witnessed Choice in Stratified Semantic Spaces","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12578","citing_title":"MARD: Mirror-Augmented Reasoning Distillation for Mechanism-Level Drug-Drug Interaction Prediction","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12731","citing_title":"Normative Robustness as a Frontier for Non-Verifiable Reasoning in LLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00341","citing_title":"DiscoLoop: Looping Discrete Embeddings and Continuous Hidden States for Multi-hop Reasoning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01811","citing_title":"\"I've Seen How This Goes\": Characterizing Diversity via Progressive Conditional Surprise","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01210","citing_title":"Can we trust LLM Self-Explanations for Entity Resolution?","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26120","citing_title":"Dynamic-dLLM: Dynamic Cache-Budget and Adaptive Parallel Decoding for Training-Free Acceleration of Diffusion LLM","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00295","citing_title":"Adaptive Order Policies for Masked Diffusion","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2405.02079","citing_title":"Argumentative Large Language Models for Explainable and Contestable Claim Verification","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15156","citing_title":"MeMo: Memory as a Model","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2406.03736","citing_title":"Your Absorbing Discrete Diffusion Secretly Models the Conditional Distributions of Clean Data","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09885","citing_title":"Diffusion-Inspired Masked Fine-Tuning for Knowledge Injection in Autoregressive LLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04943","citing_title":"The Illusion of Latent Generalization: Bi-directionality and the Reversal Curse","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15156","citing_title":"MeMo: Memory as a Model","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22773","citing_title":"Trace Mutation in Human-LLM Dialogue: The Transcript as Forensic and Mitigation Surface","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2311.05232","citing_title":"A Survey on Hallucination in Large Language Models: Principles, Taxonomy, Challenges, and Open Questions","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09678","citing_title":"Absurd World: A Simple Yet Powerful Method to Absurdify the Real-world for Probing LLM Reasoning Capabilities","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06548","citing_title":"Continuous Latent Diffusion Language Model","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2502.09992","citing_title":"Large Language Diffusion Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02442","citing_title":"Measuring AI Reasoning: A Guide for Researchers","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VQI3MMGEMTG62Y2CUXB4GT5ULX","json":"https://pith.science/pith/VQI3MMGEMTG62Y2CUXB4GT5ULX.json","graph_json":"https://pith.science/api/pith-number/VQI3MMGEMTG62Y2CUXB4GT5ULX/graph.json","events_json":"https://pith.science/api/pith-number/VQI3MMGEMTG62Y2CUXB4GT5ULX/events.json","paper":"https://pith.science/paper/VQI3MMGE"},"agent_actions":{"view_html":"https://pith.science/pith/VQI3MMGEMTG62Y2CUXB4GT5ULX","download_json":"https://pith.science/pith/VQI3MMGEMTG62Y2CUXB4GT5ULX.json","view_paper":"https://pith.science/paper/VQI3MMGE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.12288&json=true","fetch_graph":"https://pith.science/api/pith-number/VQI3MMGEMTG62Y2CUXB4GT5ULX/graph.json","fetch_events":"https://pith.science/api/pith-number/VQI3MMGEMTG62Y2CUXB4GT5ULX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VQI3MMGEMTG62Y2CUXB4GT5ULX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VQI3MMGEMTG62Y2CUXB4GT5ULX/action/storage_attestation","attest_author":"https://pith.science/pith/VQI3MMGEMTG62Y2CUXB4GT5ULX/action/author_attestation","sign_citation":"https://pith.science/pith/VQI3MMGEMTG62Y2CUXB4GT5ULX/action/citation_signature","submit_replication":"https://pith.science/pith/VQI3MMGEMTG62Y2CUXB4GT5ULX/action/replication_record"}},"created_at":"2026-07-05T08:23:12.188156+00:00","updated_at":"2026-07-05T08:23:12.188156+00:00"}