{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BHVSO6KWJQ2ZBZMBHP5EMHUKO3","short_pith_number":"pith:BHVSO6KW","schema_version":"1.0","canonical_sha256":"09eb2779564c3590e5813bfa461e8a76c725b7dff62d3a08c6fe3bf37b01ca38","source":{"kind":"arxiv","id":"2312.04350","version":3},"attestation_state":"computed","paper":{"title":"CLadder: Assessing Causal Reasoning in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bernhard Sch\\\"olkopf, Felix Leeb, Fernando Gonzalez Adauto, Kevin Blin, Luigi Gresele, Max Kleiman-Weiner, Mrinmaya Sachan, Ojasv Kamal, Yuen Chen, Zhiheng Lyu, Zhijing Jin","submitted_at":"2023-12-07T15:12:12Z","abstract_excerpt":"The ability to perform causal reasoning is widely considered a core feature of intelligence. In this work, we investigate whether large language models (LLMs) can coherently reason about causality. Much of the existing work in natural language processing (NLP) focuses on evaluating commonsense causal reasoning in LLMs, thus failing to assess whether a model can perform causal inference in accordance with a set of well-defined formal rules. To address this, we propose a new NLP task, causal inference in natural language, inspired by the \"causal inference engine\" postulated by Judea Pearl et al."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.04350","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-12-07T15:12:12Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"04eb62f1659f2a785663dc68c2e84eb3245a6fdda136bc9802619a97dd90ee3a","abstract_canon_sha256":"141ab001ab2289ed9b03b4ada5d5127b54520b7b119b07d167fb35eef1661eab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:34:36.480264Z","signature_b64":"dI9JfIUBdNHplqdVaE4YUJRSsjtGJ9+JqwY//7y8BRIvvv2vwkWXs+Afjc++KUvRFfMZbZ4My2R7DXNNXjQyAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"09eb2779564c3590e5813bfa461e8a76c725b7dff62d3a08c6fe3bf37b01ca38","last_reissued_at":"2026-07-05T07:34:36.479819Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:34:36.479819Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLadder: Assessing Causal Reasoning in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bernhard Sch\\\"olkopf, Felix Leeb, Fernando Gonzalez Adauto, Kevin Blin, Luigi Gresele, Max Kleiman-Weiner, Mrinmaya Sachan, Ojasv Kamal, Yuen Chen, Zhiheng Lyu, Zhijing Jin","submitted_at":"2023-12-07T15:12:12Z","abstract_excerpt":"The ability to perform causal reasoning is widely considered a core feature of intelligence. In this work, we investigate whether large language models (LLMs) can coherently reason about causality. Much of the existing work in natural language processing (NLP) focuses on evaluating commonsense causal reasoning in LLMs, thus failing to assess whether a model can perform causal inference in accordance with a set of well-defined formal rules. To address this, we propose a new NLP task, causal inference in natural language, inspired by the \"causal inference engine\" postulated by Judea Pearl et al."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.04350","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.04350/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.04350","created_at":"2026-07-05T07:34:36.479879+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.04350v3","created_at":"2026-07-05T07:34:36.479879+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.04350","created_at":"2026-07-05T07:34:36.479879+00:00"},{"alias_kind":"pith_short_12","alias_value":"BHVSO6KWJQ2Z","created_at":"2026-07-05T07:34:36.479879+00:00"},{"alias_kind":"pith_short_16","alias_value":"BHVSO6KWJQ2ZBZMB","created_at":"2026-07-05T07:34:36.479879+00:00"},{"alias_kind":"pith_short_8","alias_value":"BHVSO6KW","created_at":"2026-07-05T07:34:36.479879+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24370","citing_title":"When Helpfulness Overrides Causal Caution: Context-Dependent Suppression and Recovery in LLMs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04915","citing_title":"Caliper: Probing Lexical Anchors versus Causal Structure in LLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25902","citing_title":"Reading the Finetuning Prior: Verbatim Content Recovery via Contrastive Decoding Diffing","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07865","citing_title":"Instrumented data for causal scientific machine learning","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BHVSO6KWJQ2ZBZMBHP5EMHUKO3","json":"https://pith.science/pith/BHVSO6KWJQ2ZBZMBHP5EMHUKO3.json","graph_json":"https://pith.science/api/pith-number/BHVSO6KWJQ2ZBZMBHP5EMHUKO3/graph.json","events_json":"https://pith.science/api/pith-number/BHVSO6KWJQ2ZBZMBHP5EMHUKO3/events.json","paper":"https://pith.science/paper/BHVSO6KW"},"agent_actions":{"view_html":"https://pith.science/pith/BHVSO6KWJQ2ZBZMBHP5EMHUKO3","download_json":"https://pith.science/pith/BHVSO6KWJQ2ZBZMBHP5EMHUKO3.json","view_paper":"https://pith.science/paper/BHVSO6KW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.04350&json=true","fetch_graph":"https://pith.science/api/pith-number/BHVSO6KWJQ2ZBZMBHP5EMHUKO3/graph.json","fetch_events":"https://pith.science/api/pith-number/BHVSO6KWJQ2ZBZMBHP5EMHUKO3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BHVSO6KWJQ2ZBZMBHP5EMHUKO3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BHVSO6KWJQ2ZBZMBHP5EMHUKO3/action/storage_attestation","attest_author":"https://pith.science/pith/BHVSO6KWJQ2ZBZMBHP5EMHUKO3/action/author_attestation","sign_citation":"https://pith.science/pith/BHVSO6KWJQ2ZBZMBHP5EMHUKO3/action/citation_signature","submit_replication":"https://pith.science/pith/BHVSO6KWJQ2ZBZMBHP5EMHUKO3/action/replication_record"}},"created_at":"2026-07-05T07:34:36.479879+00:00","updated_at":"2026-07-05T07:34:36.479879+00:00"}