{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZR4ZRKUH5NEB2VXVPEP4KJJDVN","short_pith_number":"pith:ZR4ZRKUH","schema_version":"1.0","canonical_sha256":"cc7998aa87eb481d56f5791fc52523ab77a816b06b862e3958202562c7f865b5","source":{"kind":"arxiv","id":"2404.18681","version":1},"attestation_state":"computed","paper":{"title":"LLMClean: Context-Aware Tabular Data Cleaning via LLM-Generated OFDs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Daniel Del Gaudio, Fabian Biester, Mohamed Abdelaal","submitted_at":"2024-04-29T13:24:23Z","abstract_excerpt":"Machine learning's influence is expanding rapidly, now integral to decision-making processes from corporate strategy to the advancements in Industry 4.0. The efficacy of Artificial Intelligence broadly hinges on the caliber of data used during its training phase; optimal performance is tied to exceptional data quality. Data cleaning tools, particularly those that exploit functional dependencies within ontological frameworks or context models, are instrumental in augmenting data quality. Nevertheless, crafting these context models is a demanding task, both in terms of resources and expertise, o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.18681","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DB","submitted_at":"2024-04-29T13:24:23Z","cross_cats_sorted":[],"title_canon_sha256":"e8bb1876429fa6309e7fbbd58dc61058305b387c59c94e92befd6a293aa5d140","abstract_canon_sha256":"7f73653fa0b139f4a481f8e0c0679513018ec0b2786eab91ab26c90930d1798e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:13:19.377759Z","signature_b64":"AFH44lgPhZDWFk89YJyJj+d0PQIwO579v26iT4kPA3Z/tD4hG3YWtD37ybHayS+YWQ4jXb4u0Lc1IHXXRk0dDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc7998aa87eb481d56f5791fc52523ab77a816b06b862e3958202562c7f865b5","last_reissued_at":"2026-07-05T08:13:19.377334Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:13:19.377334Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMClean: Context-Aware Tabular Data Cleaning via LLM-Generated OFDs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Daniel Del Gaudio, Fabian Biester, Mohamed Abdelaal","submitted_at":"2024-04-29T13:24:23Z","abstract_excerpt":"Machine learning's influence is expanding rapidly, now integral to decision-making processes from corporate strategy to the advancements in Industry 4.0. The efficacy of Artificial Intelligence broadly hinges on the caliber of data used during its training phase; optimal performance is tied to exceptional data quality. Data cleaning tools, particularly those that exploit functional dependencies within ontological frameworks or context models, are instrumental in augmenting data quality. Nevertheless, crafting these context models is a demanding task, both in terms of resources and expertise, o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.18681","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.18681/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.18681","created_at":"2026-07-05T08:13:19.377399+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.18681v1","created_at":"2026-07-05T08:13:19.377399+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.18681","created_at":"2026-07-05T08:13:19.377399+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZR4ZRKUH5NEB","created_at":"2026-07-05T08:13:19.377399+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZR4ZRKUH5NEB2VXV","created_at":"2026-07-05T08:13:19.377399+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZR4ZRKUH","created_at":"2026-07-05T08:13:19.377399+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12376","citing_title":"ProfiliTable: Profiling-Driven Tabular Data Processing via Agentic Workflows","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12376","citing_title":"ProfiliTable: Profiling-Driven Tabular Data Processing via Agentic Workflows","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZR4ZRKUH5NEB2VXVPEP4KJJDVN","json":"https://pith.science/pith/ZR4ZRKUH5NEB2VXVPEP4KJJDVN.json","graph_json":"https://pith.science/api/pith-number/ZR4ZRKUH5NEB2VXVPEP4KJJDVN/graph.json","events_json":"https://pith.science/api/pith-number/ZR4ZRKUH5NEB2VXVPEP4KJJDVN/events.json","paper":"https://pith.science/paper/ZR4ZRKUH"},"agent_actions":{"view_html":"https://pith.science/pith/ZR4ZRKUH5NEB2VXVPEP4KJJDVN","download_json":"https://pith.science/pith/ZR4ZRKUH5NEB2VXVPEP4KJJDVN.json","view_paper":"https://pith.science/paper/ZR4ZRKUH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.18681&json=true","fetch_graph":"https://pith.science/api/pith-number/ZR4ZRKUH5NEB2VXVPEP4KJJDVN/graph.json","fetch_events":"https://pith.science/api/pith-number/ZR4ZRKUH5NEB2VXVPEP4KJJDVN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZR4ZRKUH5NEB2VXVPEP4KJJDVN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZR4ZRKUH5NEB2VXVPEP4KJJDVN/action/storage_attestation","attest_author":"https://pith.science/pith/ZR4ZRKUH5NEB2VXVPEP4KJJDVN/action/author_attestation","sign_citation":"https://pith.science/pith/ZR4ZRKUH5NEB2VXVPEP4KJJDVN/action/citation_signature","submit_replication":"https://pith.science/pith/ZR4ZRKUH5NEB2VXVPEP4KJJDVN/action/replication_record"}},"created_at":"2026-07-05T08:13:19.377399+00:00","updated_at":"2026-07-05T08:13:19.377399+00:00"}