{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:F3YRYDSRHSO5TG7TRUXI4O2XGI","short_pith_number":"pith:F3YRYDSR","schema_version":"1.0","canonical_sha256":"2ef11c0e513c9dd99bf38d2e8e3b57323f7389a33cd2261fb491fd67c30f70a0","source":{"kind":"arxiv","id":"2410.15547","version":1},"attestation_state":"computed","paper":{"title":"Data Cleaning Using Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Eugene Wu, Shuo Zhang, Zezhou Huang","submitted_at":"2024-10-21T00:29:40Z","abstract_excerpt":"Data cleaning is a crucial yet challenging task in data analysis, often requiring significant manual effort. To automate data cleaning, previous systems have relied on statistical rules derived from erroneous data, resulting in low accuracy and recall. This work introduces Cocoon, a novel data cleaning system that leverages large language models for rules based on semantic understanding and combines them with statistical error detection. However, data cleaning is still too complex a task for current LLMs to handle in one shot. To address this, we introduce Cocoon, which decomposes complex clea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.15547","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DB","submitted_at":"2024-10-21T00:29:40Z","cross_cats_sorted":[],"title_canon_sha256":"f20c7ca61f30a9d4574e13244d61039f40c78d8ec006ba390a19b59cbc8b9639","abstract_canon_sha256":"3f63d731f67d6f0994f7e1b6870057f6ed61458d8eb166fa60a27d2113990a64"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:23:21.118933Z","signature_b64":"ekSJ3tBQk0B5lur8D32eI47uJaHRy9MHk++mNsWlZw14N7UGIqNaeyYHS/YKxGEc3gBKMxPH4GJRE/GiL+YvCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2ef11c0e513c9dd99bf38d2e8e3b57323f7389a33cd2261fb491fd67c30f70a0","last_reissued_at":"2026-07-05T09:23:21.118479Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:23:21.118479Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Data Cleaning Using Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Eugene Wu, Shuo Zhang, Zezhou Huang","submitted_at":"2024-10-21T00:29:40Z","abstract_excerpt":"Data cleaning is a crucial yet challenging task in data analysis, often requiring significant manual effort. To automate data cleaning, previous systems have relied on statistical rules derived from erroneous data, resulting in low accuracy and recall. This work introduces Cocoon, a novel data cleaning system that leverages large language models for rules based on semantic understanding and combines them with statistical error detection. However, data cleaning is still too complex a task for current LLMs to handle in one shot. To address this, we introduce Cocoon, which decomposes complex clea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15547","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15547/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.15547","created_at":"2026-07-05T09:23:21.118536+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.15547v1","created_at":"2026-07-05T09:23:21.118536+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15547","created_at":"2026-07-05T09:23:21.118536+00:00"},{"alias_kind":"pith_short_12","alias_value":"F3YRYDSRHSO5","created_at":"2026-07-05T09:23:21.118536+00:00"},{"alias_kind":"pith_short_16","alias_value":"F3YRYDSRHSO5TG7T","created_at":"2026-07-05T09:23:21.118536+00:00"},{"alias_kind":"pith_short_8","alias_value":"F3YRYDSR","created_at":"2026-07-05T09:23:21.118536+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.12610","citing_title":"ScaleDoc: Scaling LLM-based Predicates over Large Document Collections","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21465","citing_title":"Talking Trees: Reasoning-Assisted Induction of Decision Trees for Tabular Data","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F3YRYDSRHSO5TG7TRUXI4O2XGI","json":"https://pith.science/pith/F3YRYDSRHSO5TG7TRUXI4O2XGI.json","graph_json":"https://pith.science/api/pith-number/F3YRYDSRHSO5TG7TRUXI4O2XGI/graph.json","events_json":"https://pith.science/api/pith-number/F3YRYDSRHSO5TG7TRUXI4O2XGI/events.json","paper":"https://pith.science/paper/F3YRYDSR"},"agent_actions":{"view_html":"https://pith.science/pith/F3YRYDSRHSO5TG7TRUXI4O2XGI","download_json":"https://pith.science/pith/F3YRYDSRHSO5TG7TRUXI4O2XGI.json","view_paper":"https://pith.science/paper/F3YRYDSR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.15547&json=true","fetch_graph":"https://pith.science/api/pith-number/F3YRYDSRHSO5TG7TRUXI4O2XGI/graph.json","fetch_events":"https://pith.science/api/pith-number/F3YRYDSRHSO5TG7TRUXI4O2XGI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F3YRYDSRHSO5TG7TRUXI4O2XGI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F3YRYDSRHSO5TG7TRUXI4O2XGI/action/storage_attestation","attest_author":"https://pith.science/pith/F3YRYDSRHSO5TG7TRUXI4O2XGI/action/author_attestation","sign_citation":"https://pith.science/pith/F3YRYDSRHSO5TG7TRUXI4O2XGI/action/citation_signature","submit_replication":"https://pith.science/pith/F3YRYDSRHSO5TG7TRUXI4O2XGI/action/replication_record"}},"created_at":"2026-07-05T09:23:21.118536+00:00","updated_at":"2026-07-05T09:23:21.118536+00:00"}