{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:EYPEJR3BQ47MI5YMMD7KUBUJJJ","short_pith_number":"pith:EYPEJR3B","schema_version":"1.0","canonical_sha256":"261e44c761873ec4770c60feaa06894a419b1b383677ec9ade3f70ad168adb35","source":{"kind":"arxiv","id":"2607.08093","version":1},"attestation_state":"computed","paper":{"title":"CausalDS: Benchmarking Causal Reasoning in Data-Science Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Andrej Leban, Yuekai Sun","submitted_at":"2026-07-09T04:03:26Z","abstract_excerpt":"Large language models (LLMs) increasingly act as integrated data-science agents, combining abstract reasoning with advanced tool use. Yet the relevant benchmark landscape largely divides into symbolic causal reasoning benchmarks without realistic data analysis or data analysis benchmarks without a principled causal data-generating structure. Furthermore, existing causal evaluation datasets are often restricted to curated examples from existing sources, with diversity coming from limited templatized variations rather than from systematic generation of novel synthetic causal structures. We intro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.08093","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-09T04:03:26Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"eec64ba15e9a62a83e1db684ae3610989f669ebb9c9ec9fa63c53c8b42ca165a","abstract_canon_sha256":"9345caee0f56996d0b1e552cc30492ce484c6ece55bb88239ef873c3cc63fbd6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-10T01:19:25.844928Z","signature_b64":"5C07ltnGITpmtloggHUgt+3RfIdbrY+Li43SFDFmyeQa0B9bnkAAW2tNw0V7oLtmMHk35WuvfNrVrKwguwNKDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"261e44c761873ec4770c60feaa06894a419b1b383677ec9ade3f70ad168adb35","last_reissued_at":"2026-07-10T01:19:25.844424Z","signature_status":"signed_v1","first_computed_at":"2026-07-10T01:19:25.844424Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CausalDS: Benchmarking Causal Reasoning in Data-Science Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Andrej Leban, Yuekai Sun","submitted_at":"2026-07-09T04:03:26Z","abstract_excerpt":"Large language models (LLMs) increasingly act as integrated data-science agents, combining abstract reasoning with advanced tool use. Yet the relevant benchmark landscape largely divides into symbolic causal reasoning benchmarks without realistic data analysis or data analysis benchmarks without a principled causal data-generating structure. Furthermore, existing causal evaluation datasets are often restricted to curated examples from existing sources, with diversity coming from limited templatized variations rather than from systematic generation of novel synthetic causal structures. We intro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.08093","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.08093/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.08093","created_at":"2026-07-10T01:19:25.844490+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.08093v1","created_at":"2026-07-10T01:19:25.844490+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.08093","created_at":"2026-07-10T01:19:25.844490+00:00"},{"alias_kind":"pith_short_12","alias_value":"EYPEJR3BQ47M","created_at":"2026-07-10T01:19:25.844490+00:00"},{"alias_kind":"pith_short_16","alias_value":"EYPEJR3BQ47MI5YM","created_at":"2026-07-10T01:19:25.844490+00:00"},{"alias_kind":"pith_short_8","alias_value":"EYPEJR3B","created_at":"2026-07-10T01:19:25.844490+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EYPEJR3BQ47MI5YMMD7KUBUJJJ","json":"https://pith.science/pith/EYPEJR3BQ47MI5YMMD7KUBUJJJ.json","graph_json":"https://pith.science/api/pith-number/EYPEJR3BQ47MI5YMMD7KUBUJJJ/graph.json","events_json":"https://pith.science/api/pith-number/EYPEJR3BQ47MI5YMMD7KUBUJJJ/events.json","paper":"https://pith.science/paper/EYPEJR3B"},"agent_actions":{"view_html":"https://pith.science/pith/EYPEJR3BQ47MI5YMMD7KUBUJJJ","download_json":"https://pith.science/pith/EYPEJR3BQ47MI5YMMD7KUBUJJJ.json","view_paper":"https://pith.science/paper/EYPEJR3B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.08093&json=true","fetch_graph":"https://pith.science/api/pith-number/EYPEJR3BQ47MI5YMMD7KUBUJJJ/graph.json","fetch_events":"https://pith.science/api/pith-number/EYPEJR3BQ47MI5YMMD7KUBUJJJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EYPEJR3BQ47MI5YMMD7KUBUJJJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EYPEJR3BQ47MI5YMMD7KUBUJJJ/action/storage_attestation","attest_author":"https://pith.science/pith/EYPEJR3BQ47MI5YMMD7KUBUJJJ/action/author_attestation","sign_citation":"https://pith.science/pith/EYPEJR3BQ47MI5YMMD7KUBUJJJ/action/citation_signature","submit_replication":"https://pith.science/pith/EYPEJR3BQ47MI5YMMD7KUBUJJJ/action/replication_record"}},"created_at":"2026-07-10T01:19:25.844490+00:00","updated_at":"2026-07-10T01:19:25.844490+00:00"}