{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TXFNGSPP6V6T7WHN7VRTIHNOWQ","short_pith_number":"pith:TXFNGSPP","schema_version":"1.0","canonical_sha256":"9dcad349eff57d3fd8edfd63341daeb417e2056514a81309f5f2a791d480345f","source":{"kind":"arxiv","id":"2406.07275","version":2},"attestation_state":"computed","paper":{"title":"DCA-Bench: A Benchmark for Dataset Curation Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Benhao Huang, Jiaqi Ma, Jin Huang, Xingjian Zhang, Yingzhuo Yu","submitted_at":"2024-06-11T14:02:23Z","abstract_excerpt":"The quality of datasets plays an increasingly crucial role in the research and development of modern artificial intelligence (AI). Despite the proliferation of open dataset platforms nowadays, data quality issues, such as incomplete documentation, inaccurate labels, ethical concerns, and outdated information, remain common in widely used datasets. Furthermore, these issues are often subtle and difficult to be detected by rule-based scripts, therefore requiring identification and verification by dataset users or maintainers--a process that is both time-consuming and prone to human mistakes. Wit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.07275","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-06-11T14:02:23Z","cross_cats_sorted":[],"title_canon_sha256":"0c6e363d27ed1041695e1d0f7221640e3a51e9e25c2c4f521ac352b53a38e8de","abstract_canon_sha256":"7219da7c17cf056855583e0b73b04dbe5be7d1db3e723e57a927c33bcff8ec21"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:59.399287Z","signature_b64":"R65fpgin8S8xUZdCS1qcLM6/CiqAFh6hWqGANXNgRmaNIOFZOkg3RP20c/SP1I8cFoE+Susd631doa1NHGecBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9dcad349eff57d3fd8edfd63341daeb417e2056514a81309f5f2a791d480345f","last_reissued_at":"2026-07-05T11:09:59.398773Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:59.398773Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DCA-Bench: A Benchmark for Dataset Curation Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Benhao Huang, Jiaqi Ma, Jin Huang, Xingjian Zhang, Yingzhuo Yu","submitted_at":"2024-06-11T14:02:23Z","abstract_excerpt":"The quality of datasets plays an increasingly crucial role in the research and development of modern artificial intelligence (AI). Despite the proliferation of open dataset platforms nowadays, data quality issues, such as incomplete documentation, inaccurate labels, ethical concerns, and outdated information, remain common in widely used datasets. Furthermore, these issues are often subtle and difficult to be detected by rule-based scripts, therefore requiring identification and verification by dataset users or maintainers--a process that is both time-consuming and prone to human mistakes. Wit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.07275","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.07275/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.07275","created_at":"2026-07-05T11:09:59.398833+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.07275v2","created_at":"2026-07-05T11:09:59.398833+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.07275","created_at":"2026-07-05T11:09:59.398833+00:00"},{"alias_kind":"pith_short_12","alias_value":"TXFNGSPP6V6T","created_at":"2026-07-05T11:09:59.398833+00:00"},{"alias_kind":"pith_short_16","alias_value":"TXFNGSPP6V6T7WHN","created_at":"2026-07-05T11:09:59.398833+00:00"},{"alias_kind":"pith_short_8","alias_value":"TXFNGSPP","created_at":"2026-07-05T11:09:59.398833+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.21460","citing_title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","ref_index":141,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TXFNGSPP6V6T7WHN7VRTIHNOWQ","json":"https://pith.science/pith/TXFNGSPP6V6T7WHN7VRTIHNOWQ.json","graph_json":"https://pith.science/api/pith-number/TXFNGSPP6V6T7WHN7VRTIHNOWQ/graph.json","events_json":"https://pith.science/api/pith-number/TXFNGSPP6V6T7WHN7VRTIHNOWQ/events.json","paper":"https://pith.science/paper/TXFNGSPP"},"agent_actions":{"view_html":"https://pith.science/pith/TXFNGSPP6V6T7WHN7VRTIHNOWQ","download_json":"https://pith.science/pith/TXFNGSPP6V6T7WHN7VRTIHNOWQ.json","view_paper":"https://pith.science/paper/TXFNGSPP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.07275&json=true","fetch_graph":"https://pith.science/api/pith-number/TXFNGSPP6V6T7WHN7VRTIHNOWQ/graph.json","fetch_events":"https://pith.science/api/pith-number/TXFNGSPP6V6T7WHN7VRTIHNOWQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TXFNGSPP6V6T7WHN7VRTIHNOWQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TXFNGSPP6V6T7WHN7VRTIHNOWQ/action/storage_attestation","attest_author":"https://pith.science/pith/TXFNGSPP6V6T7WHN7VRTIHNOWQ/action/author_attestation","sign_citation":"https://pith.science/pith/TXFNGSPP6V6T7WHN7VRTIHNOWQ/action/citation_signature","submit_replication":"https://pith.science/pith/TXFNGSPP6V6T7WHN7VRTIHNOWQ/action/replication_record"}},"created_at":"2026-07-05T11:09:59.398833+00:00","updated_at":"2026-07-05T11:09:59.398833+00:00"}