{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KWMJ3EXJVJIY5QTB3GG3FDVCNN","short_pith_number":"pith:KWMJ3EXJ","schema_version":"1.0","canonical_sha256":"55989d92e9aa518ec261d98db28ea26b7490223c303298a6192dc10711f9a657","source":{"kind":"arxiv","id":"2410.16676","version":4},"attestation_state":"computed","paper":{"title":"CausalEval: Towards Better Causal Reasoning in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Dawei Li, Delin Chen, Liangming Pan, Longxuan Yu, Qingyang Wu, Qingzhen Liu, Siheng Xiong, Xiaoze Liu, Zhikai Chen","submitted_at":"2024-10-22T04:18:19Z","abstract_excerpt":"Causal reasoning (CR) is a crucial aspect of intelligence, essential for problem-solving, decision-making, and understanding the world. While language models (LMs) can generate rationales for their outputs, their ability to reliably perform causal reasoning remains uncertain, often falling short in tasks requiring a deep understanding of causality. In this paper, we introduce CausalEval, a comprehensive review of research aimed at enhancing LMs for causal reasoning, coupled with an empirical evaluation of current models and methods. We categorize existing methods based on the role of LMs: eith"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.16676","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-10-22T04:18:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"1bfc7a1bf34bd50e4a34011d18512ead52f19cebd0ba4bafeeb1be16405fe99a","abstract_canon_sha256":"3910afb325050e66c6625ba79974aa7048b9a02d666ce1f051b6f3b20a1801ce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:46.699723Z","signature_b64":"Wy8sRhyrgc+9ZSUhy6FaRRBjVl76sSYshsImNHXYqD3YkX2dQDQGrrVzQ27A9bk6WGACg0Zuqw6/T+LzkPTgDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"55989d92e9aa518ec261d98db28ea26b7490223c303298a6192dc10711f9a657","last_reissued_at":"2026-07-05T10:15:46.699083Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:46.699083Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CausalEval: Towards Better Causal Reasoning in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Dawei Li, Delin Chen, Liangming Pan, Longxuan Yu, Qingyang Wu, Qingzhen Liu, Siheng Xiong, Xiaoze Liu, Zhikai Chen","submitted_at":"2024-10-22T04:18:19Z","abstract_excerpt":"Causal reasoning (CR) is a crucial aspect of intelligence, essential for problem-solving, decision-making, and understanding the world. While language models (LMs) can generate rationales for their outputs, their ability to reliably perform causal reasoning remains uncertain, often falling short in tasks requiring a deep understanding of causality. In this paper, we introduce CausalEval, a comprehensive review of research aimed at enhancing LMs for causal reasoning, coupled with an empirical evaluation of current models and methods. We categorize existing methods based on the role of LMs: eith"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.16676","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.16676/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.16676","created_at":"2026-07-05T10:15:46.699191+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.16676v4","created_at":"2026-07-05T10:15:46.699191+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.16676","created_at":"2026-07-05T10:15:46.699191+00:00"},{"alias_kind":"pith_short_12","alias_value":"KWMJ3EXJVJIY","created_at":"2026-07-05T10:15:46.699191+00:00"},{"alias_kind":"pith_short_16","alias_value":"KWMJ3EXJVJIY5QTB","created_at":"2026-07-05T10:15:46.699191+00:00"},{"alias_kind":"pith_short_8","alias_value":"KWMJ3EXJ","created_at":"2026-07-05T10:15:46.699191+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23414","citing_title":"When Planning Fails Despite Correct Execution: On Epistemic Calibration for LLM-Based Multi-Agent Systems","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":216,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KWMJ3EXJVJIY5QTB3GG3FDVCNN","json":"https://pith.science/pith/KWMJ3EXJVJIY5QTB3GG3FDVCNN.json","graph_json":"https://pith.science/api/pith-number/KWMJ3EXJVJIY5QTB3GG3FDVCNN/graph.json","events_json":"https://pith.science/api/pith-number/KWMJ3EXJVJIY5QTB3GG3FDVCNN/events.json","paper":"https://pith.science/paper/KWMJ3EXJ"},"agent_actions":{"view_html":"https://pith.science/pith/KWMJ3EXJVJIY5QTB3GG3FDVCNN","download_json":"https://pith.science/pith/KWMJ3EXJVJIY5QTB3GG3FDVCNN.json","view_paper":"https://pith.science/paper/KWMJ3EXJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.16676&json=true","fetch_graph":"https://pith.science/api/pith-number/KWMJ3EXJVJIY5QTB3GG3FDVCNN/graph.json","fetch_events":"https://pith.science/api/pith-number/KWMJ3EXJVJIY5QTB3GG3FDVCNN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KWMJ3EXJVJIY5QTB3GG3FDVCNN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KWMJ3EXJVJIY5QTB3GG3FDVCNN/action/storage_attestation","attest_author":"https://pith.science/pith/KWMJ3EXJVJIY5QTB3GG3FDVCNN/action/author_attestation","sign_citation":"https://pith.science/pith/KWMJ3EXJVJIY5QTB3GG3FDVCNN/action/citation_signature","submit_replication":"https://pith.science/pith/KWMJ3EXJVJIY5QTB3GG3FDVCNN/action/replication_record"}},"created_at":"2026-07-05T10:15:46.699191+00:00","updated_at":"2026-07-05T10:15:46.699191+00:00"}