{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PTT7U4IEKLPTDFR7QQQAJXA6PL","short_pith_number":"pith:PTT7U4IE","schema_version":"1.0","canonical_sha256":"7ce7fa710452df31963f842004dc1e7ad3d01e0a7fbab933a32ee56bdbf15112","source":{"kind":"arxiv","id":"2411.11908","version":1},"attestation_state":"computed","paper":{"title":"LLM4DS: Evaluating Large Language Models for Data Science Code Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.ET"],"primary_cat":"cs.SE","authors_text":"Everton Guimaraes, Nathalia Nascimento, Sai Sanjna Chintakunta, Santhosh Anitha Boominathan","submitted_at":"2024-11-16T18:43:26Z","abstract_excerpt":"The adoption of Large Language Models (LLMs) for code generation in data science offers substantial potential for enhancing tasks such as data manipulation, statistical analysis, and visualization. However, the effectiveness of these models in the data science domain remains underexplored. This paper presents a controlled experiment that empirically assesses the performance of four leading LLM-based AI assistants-Microsoft Copilot (GPT-4 Turbo), ChatGPT (o1-preview), Claude (3.5 Sonnet), and Perplexity Labs (Llama-3.1-70b-instruct)-on a diverse set of data science coding challenges sourced fro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.11908","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-11-16T18:43:26Z","cross_cats_sorted":["cs.AI","cs.ET"],"title_canon_sha256":"074976828f3b08ee0791136ccd64616569946cb3697df286e8b99fd425d83562","abstract_canon_sha256":"700e66f18e359704ab954dc3c11e8a613daa5ae6a7f751a9e63ede7864bea85a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:37:13.864947Z","signature_b64":"+YfEJQVT6NDs+Ue2FX4/v6kLoJ1UWYAs8c0SsMZkjj674MabA6RU6WWxEr4GWrbNZQw+4pzYsf3N11BXSp/ADg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7ce7fa710452df31963f842004dc1e7ad3d01e0a7fbab933a32ee56bdbf15112","last_reissued_at":"2026-07-05T09:37:13.864483Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:37:13.864483Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM4DS: Evaluating Large Language Models for Data Science Code Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.ET"],"primary_cat":"cs.SE","authors_text":"Everton Guimaraes, Nathalia Nascimento, Sai Sanjna Chintakunta, Santhosh Anitha Boominathan","submitted_at":"2024-11-16T18:43:26Z","abstract_excerpt":"The adoption of Large Language Models (LLMs) for code generation in data science offers substantial potential for enhancing tasks such as data manipulation, statistical analysis, and visualization. However, the effectiveness of these models in the data science domain remains underexplored. This paper presents a controlled experiment that empirically assesses the performance of four leading LLM-based AI assistants-Microsoft Copilot (GPT-4 Turbo), ChatGPT (o1-preview), Claude (3.5 Sonnet), and Perplexity Labs (Llama-3.1-70b-instruct)-on a diverse set of data science coding challenges sourced fro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.11908","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.11908/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.11908","created_at":"2026-07-05T09:37:13.864536+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.11908v1","created_at":"2026-07-05T09:37:13.864536+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.11908","created_at":"2026-07-05T09:37:13.864536+00:00"},{"alias_kind":"pith_short_12","alias_value":"PTT7U4IEKLPT","created_at":"2026-07-05T09:37:13.864536+00:00"},{"alias_kind":"pith_short_16","alias_value":"PTT7U4IEKLPTDFR7","created_at":"2026-07-05T09:37:13.864536+00:00"},{"alias_kind":"pith_short_8","alias_value":"PTT7U4IE","created_at":"2026-07-05T09:37:13.864536+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07504","citing_title":"Do LLM-Generated Skills Make Better AI Data Scientists? A Component Ablation Across Data-Science Workflows","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PTT7U4IEKLPTDFR7QQQAJXA6PL","json":"https://pith.science/pith/PTT7U4IEKLPTDFR7QQQAJXA6PL.json","graph_json":"https://pith.science/api/pith-number/PTT7U4IEKLPTDFR7QQQAJXA6PL/graph.json","events_json":"https://pith.science/api/pith-number/PTT7U4IEKLPTDFR7QQQAJXA6PL/events.json","paper":"https://pith.science/paper/PTT7U4IE"},"agent_actions":{"view_html":"https://pith.science/pith/PTT7U4IEKLPTDFR7QQQAJXA6PL","download_json":"https://pith.science/pith/PTT7U4IEKLPTDFR7QQQAJXA6PL.json","view_paper":"https://pith.science/paper/PTT7U4IE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.11908&json=true","fetch_graph":"https://pith.science/api/pith-number/PTT7U4IEKLPTDFR7QQQAJXA6PL/graph.json","fetch_events":"https://pith.science/api/pith-number/PTT7U4IEKLPTDFR7QQQAJXA6PL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PTT7U4IEKLPTDFR7QQQAJXA6PL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PTT7U4IEKLPTDFR7QQQAJXA6PL/action/storage_attestation","attest_author":"https://pith.science/pith/PTT7U4IEKLPTDFR7QQQAJXA6PL/action/author_attestation","sign_citation":"https://pith.science/pith/PTT7U4IEKLPTDFR7QQQAJXA6PL/action/citation_signature","submit_replication":"https://pith.science/pith/PTT7U4IEKLPTDFR7QQQAJXA6PL/action/replication_record"}},"created_at":"2026-07-05T09:37:13.864536+00:00","updated_at":"2026-07-05T09:37:13.864536+00:00"}