{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OF65BMU5ANY6VJD6UG7ZRYMR3S","short_pith_number":"pith:OF65BMU5","schema_version":"1.0","canonical_sha256":"717dd0b29d0371eaa47ea1bf98e191dc891034518681496062bd37d3c2cf0ac7","source":{"kind":"arxiv","id":"2502.10215","version":2},"attestation_state":"computed","paper":{"title":"Do Large Language Models Reason Causally Like Us? Even Better?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Bob Rehder, Brenden M. Lake, Charley M. Wu, Hanna M. Dettki","submitted_at":"2025-02-14T15:09:15Z","abstract_excerpt":"Causal reasoning is a core component of intelligence. Large language models (LLMs) have shown impressive capabilities in generating human-like text, raising questions about whether their responses reflect true understanding or statistical patterns. We compared causal reasoning in humans and four LLMs using tasks based on collider graphs, rating the likelihood of a query variable occurring given evidence from other variables. LLMs' causal inferences ranged from often nonsensical (GPT-3.5) to human-like to often more normatively aligned than those of humans (GPT-4o, Gemini-Pro, and Claude). Comp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.10215","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-02-14T15:09:15Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"c644d5989487c5721021409e84a701697756739d97281a47c163b425a043d55a","abstract_canon_sha256":"8407cc5dfa85c43f9cc16483405f566f2a821da4f71d9b1d920641fb3f2a245a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:23.561409Z","signature_b64":"TBbisrtbu85axJt4T85NwoXXSgGtTKuvxTXe9j1rPwyJ5wpa7PVWY2TMyonNXhDjtn0lpJ+It011PPrSdETcDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"717dd0b29d0371eaa47ea1bf98e191dc891034518681496062bd37d3c2cf0ac7","last_reissued_at":"2026-07-05T11:17:23.560931Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:23.560931Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Large Language Models Reason Causally Like Us? Even Better?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Bob Rehder, Brenden M. Lake, Charley M. Wu, Hanna M. Dettki","submitted_at":"2025-02-14T15:09:15Z","abstract_excerpt":"Causal reasoning is a core component of intelligence. Large language models (LLMs) have shown impressive capabilities in generating human-like text, raising questions about whether their responses reflect true understanding or statistical patterns. We compared causal reasoning in humans and four LLMs using tasks based on collider graphs, rating the likelihood of a query variable occurring given evidence from other variables. LLMs' causal inferences ranged from often nonsensical (GPT-3.5) to human-like to often more normatively aligned than those of humans (GPT-4o, Gemini-Pro, and Claude). Comp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.10215","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.10215/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.10215","created_at":"2026-07-05T11:17:23.560983+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.10215v2","created_at":"2026-07-05T11:17:23.560983+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.10215","created_at":"2026-07-05T11:17:23.560983+00:00"},{"alias_kind":"pith_short_12","alias_value":"OF65BMU5ANY6","created_at":"2026-07-05T11:17:23.560983+00:00"},{"alias_kind":"pith_short_16","alias_value":"OF65BMU5ANY6VJD6","created_at":"2026-07-05T11:17:23.560983+00:00"},{"alias_kind":"pith_short_8","alias_value":"OF65BMU5","created_at":"2026-07-05T11:17:23.560983+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OF65BMU5ANY6VJD6UG7ZRYMR3S","json":"https://pith.science/pith/OF65BMU5ANY6VJD6UG7ZRYMR3S.json","graph_json":"https://pith.science/api/pith-number/OF65BMU5ANY6VJD6UG7ZRYMR3S/graph.json","events_json":"https://pith.science/api/pith-number/OF65BMU5ANY6VJD6UG7ZRYMR3S/events.json","paper":"https://pith.science/paper/OF65BMU5"},"agent_actions":{"view_html":"https://pith.science/pith/OF65BMU5ANY6VJD6UG7ZRYMR3S","download_json":"https://pith.science/pith/OF65BMU5ANY6VJD6UG7ZRYMR3S.json","view_paper":"https://pith.science/paper/OF65BMU5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.10215&json=true","fetch_graph":"https://pith.science/api/pith-number/OF65BMU5ANY6VJD6UG7ZRYMR3S/graph.json","fetch_events":"https://pith.science/api/pith-number/OF65BMU5ANY6VJD6UG7ZRYMR3S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OF65BMU5ANY6VJD6UG7ZRYMR3S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OF65BMU5ANY6VJD6UG7ZRYMR3S/action/storage_attestation","attest_author":"https://pith.science/pith/OF65BMU5ANY6VJD6UG7ZRYMR3S/action/author_attestation","sign_citation":"https://pith.science/pith/OF65BMU5ANY6VJD6UG7ZRYMR3S/action/citation_signature","submit_replication":"https://pith.science/pith/OF65BMU5ANY6VJD6UG7ZRYMR3S/action/replication_record"}},"created_at":"2026-07-05T11:17:23.560983+00:00","updated_at":"2026-07-05T11:17:23.560983+00:00"}