{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4YNU4AMOZV2SM4H57QLGBEI2BK","short_pith_number":"pith:4YNU4AMO","schema_version":"1.0","canonical_sha256":"e61b4e018ecd752670fdfc1660911a0aa823105007c231dfd105ca68a8735ff1","source":{"kind":"arxiv","id":"2404.15320","version":2},"attestation_state":"computed","paper":{"title":"Using Large Language Models to Enrich the Documentation of Datasets for Machine Learning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.DL","authors_text":"Abel G\\'omez, Joan Giner-Miguelez, Jordi Cabot","submitted_at":"2024-04-04T10:09:28Z","abstract_excerpt":"Recent regulatory initiatives like the European AI Act and relevant voices in the Machine Learning (ML) community stress the need to describe datasets along several key dimensions for trustworthy AI, such as the provenance processes and social concerns. However, this information is typically presented as unstructured text in accompanying documentation, hampering their automated analysis and processing. In this work, we explore using large language models (LLM) and a set of prompting strategies to automatically extract these dimensions from documents and enrich the dataset description with them"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.15320","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.DL","submitted_at":"2024-04-04T10:09:28Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"e6cd4fb878afaf7f5e49234313c991a9bc006d46262ced7915d71287e0641e03","abstract_canon_sha256":"42ab23132925aafe7d8d7bf51932ac3504cf3f729b0c92483e61f46741aa13ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:39.782748Z","signature_b64":"IECYGE4PCnBhxJ/rXhSthEbZvM376oedZshcdKSeOdQHStHqnNGylUdFrB94mc7uU5bzgISFPgyQITB4WDdvDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e61b4e018ecd752670fdfc1660911a0aa823105007c231dfd105ca68a8735ff1","last_reissued_at":"2026-07-05T08:22:39.782259Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:39.782259Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Using Large Language Models to Enrich the Documentation of Datasets for Machine Learning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.DL","authors_text":"Abel G\\'omez, Joan Giner-Miguelez, Jordi Cabot","submitted_at":"2024-04-04T10:09:28Z","abstract_excerpt":"Recent regulatory initiatives like the European AI Act and relevant voices in the Machine Learning (ML) community stress the need to describe datasets along several key dimensions for trustworthy AI, such as the provenance processes and social concerns. However, this information is typically presented as unstructured text in accompanying documentation, hampering their automated analysis and processing. In this work, we explore using large language models (LLM) and a set of prompting strategies to automatically extract these dimensions from documents and enrich the dataset description with them"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.15320","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.15320/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.15320","created_at":"2026-07-05T08:22:39.782313+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.15320v2","created_at":"2026-07-05T08:22:39.782313+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.15320","created_at":"2026-07-05T08:22:39.782313+00:00"},{"alias_kind":"pith_short_12","alias_value":"4YNU4AMOZV2S","created_at":"2026-07-05T08:22:39.782313+00:00"},{"alias_kind":"pith_short_16","alias_value":"4YNU4AMOZV2SM4H5","created_at":"2026-07-05T08:22:39.782313+00:00"},{"alias_kind":"pith_short_8","alias_value":"4YNU4AMO","created_at":"2026-07-05T08:22:39.782313+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.06688","citing_title":"Network Intrusion Datasets: A Survey, Limitations, and Recommendations","ref_index":81,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4YNU4AMOZV2SM4H57QLGBEI2BK","json":"https://pith.science/pith/4YNU4AMOZV2SM4H57QLGBEI2BK.json","graph_json":"https://pith.science/api/pith-number/4YNU4AMOZV2SM4H57QLGBEI2BK/graph.json","events_json":"https://pith.science/api/pith-number/4YNU4AMOZV2SM4H57QLGBEI2BK/events.json","paper":"https://pith.science/paper/4YNU4AMO"},"agent_actions":{"view_html":"https://pith.science/pith/4YNU4AMOZV2SM4H57QLGBEI2BK","download_json":"https://pith.science/pith/4YNU4AMOZV2SM4H57QLGBEI2BK.json","view_paper":"https://pith.science/paper/4YNU4AMO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.15320&json=true","fetch_graph":"https://pith.science/api/pith-number/4YNU4AMOZV2SM4H57QLGBEI2BK/graph.json","fetch_events":"https://pith.science/api/pith-number/4YNU4AMOZV2SM4H57QLGBEI2BK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4YNU4AMOZV2SM4H57QLGBEI2BK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4YNU4AMOZV2SM4H57QLGBEI2BK/action/storage_attestation","attest_author":"https://pith.science/pith/4YNU4AMOZV2SM4H57QLGBEI2BK/action/author_attestation","sign_citation":"https://pith.science/pith/4YNU4AMOZV2SM4H57QLGBEI2BK/action/citation_signature","submit_replication":"https://pith.science/pith/4YNU4AMOZV2SM4H57QLGBEI2BK/action/replication_record"}},"created_at":"2026-07-05T08:22:39.782313+00:00","updated_at":"2026-07-05T08:22:39.782313+00:00"}