{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GKDRCX2KU4UXO33O5N65J4RHKQ","short_pith_number":"pith:GKDRCX2K","schema_version":"1.0","canonical_sha256":"3287115f4aa729776f6eeb7dd4f227543375d1b49ae0536a05c565362bca8aa7","source":{"kind":"arxiv","id":"2503.06664","version":1},"attestation_state":"computed","paper":{"title":"Exploring LLM Agents for Cleaning Tabular Machine Learning Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Artur Dox, Christian Holz, Tommaso Bendinelli","submitted_at":"2025-03-09T15:29:46Z","abstract_excerpt":"High-quality, error-free datasets are a key ingredient in building reliable, accurate, and unbiased machine learning (ML) models. However, real world datasets often suffer from errors due to sensor malfunctions, data entry mistakes, or improper data integration across multiple sources that can severely degrade model performance. Detecting and correcting these issues typically require tailor-made solutions and demand extensive domain expertise. Consequently, automation is challenging, rendering the process labor-intensive and tedious. In this study, we investigate whether Large Language Models "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.06664","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-09T15:29:46Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"67e2e4a522fea8454a8a4b521c5b229ccb1353e9739958dcf82946b4e5f25e7a","abstract_canon_sha256":"16792f2ddd1f3f55a0015c4dff707f32782d1f5a03bfc0e7872fb75c8e738936"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:27:39.705286Z","signature_b64":"wocro7y7adpE5XKDWbbrP96Rv6FoyHrSkAEr69p1+jVk0CFBlhup8nW3/X38hBI2inR99XStVG+h3szNWy7wCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3287115f4aa729776f6eeb7dd4f227543375d1b49ae0536a05c565362bca8aa7","last_reissued_at":"2026-07-05T10:27:39.704755Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:27:39.704755Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring LLM Agents for Cleaning Tabular Machine Learning Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Artur Dox, Christian Holz, Tommaso Bendinelli","submitted_at":"2025-03-09T15:29:46Z","abstract_excerpt":"High-quality, error-free datasets are a key ingredient in building reliable, accurate, and unbiased machine learning (ML) models. However, real world datasets often suffer from errors due to sensor malfunctions, data entry mistakes, or improper data integration across multiple sources that can severely degrade model performance. Detecting and correcting these issues typically require tailor-made solutions and demand extensive domain expertise. Consequently, automation is challenging, rendering the process labor-intensive and tedious. In this study, we investigate whether Large Language Models "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.06664","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.06664/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.06664","created_at":"2026-07-05T10:27:39.704815+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.06664v1","created_at":"2026-07-05T10:27:39.704815+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.06664","created_at":"2026-07-05T10:27:39.704815+00:00"},{"alias_kind":"pith_short_12","alias_value":"GKDRCX2KU4UX","created_at":"2026-07-05T10:27:39.704815+00:00"},{"alias_kind":"pith_short_16","alias_value":"GKDRCX2KU4UXO33O","created_at":"2026-07-05T10:27:39.704815+00:00"},{"alias_kind":"pith_short_8","alias_value":"GKDRCX2K","created_at":"2026-07-05T10:27:39.704815+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17915","citing_title":"Trustworthy Self-Composable Big-Data-as-a-Service: An LLM-Orchestrated Multi-Agent Framework for Automated Data Engineering, AutoML, MLOps Deployment, and Drift-Aware Lifecycle Optimization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12376","citing_title":"ProfiliTable: Profiling-Driven Tabular Data Processing via Agentic Workflows","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21465","citing_title":"Talking Trees: Reasoning-Assisted Induction of Decision Trees for Tabular Data","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12376","citing_title":"ProfiliTable: Profiling-Driven Tabular Data Processing via Agentic Workflows","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GKDRCX2KU4UXO33O5N65J4RHKQ","json":"https://pith.science/pith/GKDRCX2KU4UXO33O5N65J4RHKQ.json","graph_json":"https://pith.science/api/pith-number/GKDRCX2KU4UXO33O5N65J4RHKQ/graph.json","events_json":"https://pith.science/api/pith-number/GKDRCX2KU4UXO33O5N65J4RHKQ/events.json","paper":"https://pith.science/paper/GKDRCX2K"},"agent_actions":{"view_html":"https://pith.science/pith/GKDRCX2KU4UXO33O5N65J4RHKQ","download_json":"https://pith.science/pith/GKDRCX2KU4UXO33O5N65J4RHKQ.json","view_paper":"https://pith.science/paper/GKDRCX2K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.06664&json=true","fetch_graph":"https://pith.science/api/pith-number/GKDRCX2KU4UXO33O5N65J4RHKQ/graph.json","fetch_events":"https://pith.science/api/pith-number/GKDRCX2KU4UXO33O5N65J4RHKQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GKDRCX2KU4UXO33O5N65J4RHKQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GKDRCX2KU4UXO33O5N65J4RHKQ/action/storage_attestation","attest_author":"https://pith.science/pith/GKDRCX2KU4UXO33O5N65J4RHKQ/action/author_attestation","sign_citation":"https://pith.science/pith/GKDRCX2KU4UXO33O5N65J4RHKQ/action/citation_signature","submit_replication":"https://pith.science/pith/GKDRCX2KU4UXO33O5N65J4RHKQ/action/replication_record"}},"created_at":"2026-07-05T10:27:39.704815+00:00","updated_at":"2026-07-05T10:27:39.704815+00:00"}