{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NSGWMCMRHRUO6QDWZDQX4AENGK","short_pith_number":"pith:NSGWMCMR","schema_version":"1.0","canonical_sha256":"6c8d6609913c68ef4076c8e17e008d32bdb5565c4aff8c5b884c4b20ed2a651d","source":{"kind":"arxiv","id":"2403.08291","version":4},"attestation_state":"computed","paper":{"title":"CleanAgent: Automating Data Standardization with LLM-based Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Danrui Qi, Jiannan Wang, Zhengjie Miao","submitted_at":"2024-03-13T06:54:15Z","abstract_excerpt":"Data standardization is a crucial part of the data science life cycle. While tools like Pandas offer robust functionalities, their complexity and the manual effort required for customizing code to diverse column types pose significant challenges. Although large language models (LLMs) like ChatGPT have shown promise in automating this process through natural language understanding and code generation, it still demands expert-level programming knowledge and continuous interaction for prompt refinement. To solve these challenges, our key idea is to propose a Python library with declarative, unifi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.08291","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-03-13T06:54:15Z","cross_cats_sorted":["cs.AI","cs.MA"],"title_canon_sha256":"2b8a085f9ffb2e95aeb1b168b08b253626fb096307663ccfb09d4139ca9a6ada","abstract_canon_sha256":"d548d12ae2b7555079b24431abee42464325fd6e7fc8057691a22b209c156d03"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:55.465738Z","signature_b64":"kBntocVqB4rRkrvdyOiM/AsY/GeykOTk4DMIlgahQ9WHgybV0dkcFw9Qc8ACU0PHv+j06QI2lwvzqvjyyXXPAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6c8d6609913c68ef4076c8e17e008d32bdb5565c4aff8c5b884c4b20ed2a651d","last_reissued_at":"2026-07-05T11:13:55.465151Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:55.465151Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CleanAgent: Automating Data Standardization with LLM-based Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Danrui Qi, Jiannan Wang, Zhengjie Miao","submitted_at":"2024-03-13T06:54:15Z","abstract_excerpt":"Data standardization is a crucial part of the data science life cycle. While tools like Pandas offer robust functionalities, their complexity and the manual effort required for customizing code to diverse column types pose significant challenges. Although large language models (LLMs) like ChatGPT have shown promise in automating this process through natural language understanding and code generation, it still demands expert-level programming knowledge and continuous interaction for prompt refinement. To solve these challenges, our key idea is to propose a Python library with declarative, unifi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.08291","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.08291/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.08291","created_at":"2026-07-05T11:13:55.465212+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.08291v4","created_at":"2026-07-05T11:13:55.465212+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.08291","created_at":"2026-07-05T11:13:55.465212+00:00"},{"alias_kind":"pith_short_12","alias_value":"NSGWMCMRHRUO","created_at":"2026-07-05T11:13:55.465212+00:00"},{"alias_kind":"pith_short_16","alias_value":"NSGWMCMRHRUO6QDW","created_at":"2026-07-05T11:13:55.465212+00:00"},{"alias_kind":"pith_short_8","alias_value":"NSGWMCMR","created_at":"2026-07-05T11:13:55.465212+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25388","citing_title":"TabClean: Reusable LLM-Synthesized Programs for Tabular Data Cleaning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02866","citing_title":"When Helping Hurts and How to Fix It: Multi-Agent Debate for Data Cleaning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12376","citing_title":"ProfiliTable: Profiling-Driven Tabular Data Processing via Agentic Workflows","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12376","citing_title":"ProfiliTable: Profiling-Driven Tabular Data Processing via Agentic Workflows","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NSGWMCMRHRUO6QDWZDQX4AENGK","json":"https://pith.science/pith/NSGWMCMRHRUO6QDWZDQX4AENGK.json","graph_json":"https://pith.science/api/pith-number/NSGWMCMRHRUO6QDWZDQX4AENGK/graph.json","events_json":"https://pith.science/api/pith-number/NSGWMCMRHRUO6QDWZDQX4AENGK/events.json","paper":"https://pith.science/paper/NSGWMCMR"},"agent_actions":{"view_html":"https://pith.science/pith/NSGWMCMRHRUO6QDWZDQX4AENGK","download_json":"https://pith.science/pith/NSGWMCMRHRUO6QDWZDQX4AENGK.json","view_paper":"https://pith.science/paper/NSGWMCMR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.08291&json=true","fetch_graph":"https://pith.science/api/pith-number/NSGWMCMRHRUO6QDWZDQX4AENGK/graph.json","fetch_events":"https://pith.science/api/pith-number/NSGWMCMRHRUO6QDWZDQX4AENGK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NSGWMCMRHRUO6QDWZDQX4AENGK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NSGWMCMRHRUO6QDWZDQX4AENGK/action/storage_attestation","attest_author":"https://pith.science/pith/NSGWMCMRHRUO6QDWZDQX4AENGK/action/author_attestation","sign_citation":"https://pith.science/pith/NSGWMCMRHRUO6QDWZDQX4AENGK/action/citation_signature","submit_replication":"https://pith.science/pith/NSGWMCMRHRUO6QDWZDQX4AENGK/action/replication_record"}},"created_at":"2026-07-05T11:13:55.465212+00:00","updated_at":"2026-07-05T11:13:55.465212+00:00"}