{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TDGKLRGH5PPVBW4Y7ABBOSFRBX","short_pith_number":"pith:TDGKLRGH","schema_version":"1.0","canonical_sha256":"98cca5c4c7ebdf50db98f8021748b10def8d7245339b722e6c952b70278cdbc6","source":{"kind":"arxiv","id":"2301.05948","version":3},"attestation_state":"computed","paper":{"title":"tasksource: A Dataset Harmonization Framework for Streamlined NLP Multi-Task Learning and Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Damien Sileo","submitted_at":"2023-01-14T16:38:04Z","abstract_excerpt":"The HuggingFace Datasets Hub hosts thousands of datasets, offering exciting opportunities for language model training and evaluation. However, datasets for a specific task type often have different schemas, making harmonization challenging. Multi-task training or evaluation necessitates manual work to fit data into task templates. Several initiatives independently tackle this issue by releasing harmonized datasets or providing harmonization codes to preprocess datasets into a consistent format. We identify patterns across previous preprocessing efforts, such as column name mapping and extracti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.05948","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-01-14T16:38:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4fdb569f71fe677c8ec55d45ed2f28e5c1790c7514d6492680e301116235e3aa","abstract_canon_sha256":"34bd448adf5d4a1cf77a9579bb8ffe97e9338e8d6e4a3562b86c466f402abd10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:10:28.043840Z","signature_b64":"vI0REVsrFNLq9WldkkStguYxHiUacxnRhqZoZuUSt+KFwsAwZX+a04e9Z+/m6L0HoFFHV6Q1c7l76KX9Dyv3CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"98cca5c4c7ebdf50db98f8021748b10def8d7245339b722e6c952b70278cdbc6","last_reissued_at":"2026-07-05T06:10:28.043400Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:10:28.043400Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"tasksource: A Dataset Harmonization Framework for Streamlined NLP Multi-Task Learning and Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Damien Sileo","submitted_at":"2023-01-14T16:38:04Z","abstract_excerpt":"The HuggingFace Datasets Hub hosts thousands of datasets, offering exciting opportunities for language model training and evaluation. However, datasets for a specific task type often have different schemas, making harmonization challenging. Multi-task training or evaluation necessitates manual work to fit data into task templates. Several initiatives independently tackle this issue by releasing harmonized datasets or providing harmonization codes to preprocess datasets into a consistent format. We identify patterns across previous preprocessing efforts, such as column name mapping and extracti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.05948","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.05948/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.05948","created_at":"2026-07-05T06:10:28.043457+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.05948v3","created_at":"2026-07-05T06:10:28.043457+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.05948","created_at":"2026-07-05T06:10:28.043457+00:00"},{"alias_kind":"pith_short_12","alias_value":"TDGKLRGH5PPV","created_at":"2026-07-05T06:10:28.043457+00:00"},{"alias_kind":"pith_short_16","alias_value":"TDGKLRGH5PPVBW4Y","created_at":"2026-07-05T06:10:28.043457+00:00"},{"alias_kind":"pith_short_8","alias_value":"TDGKLRGH","created_at":"2026-07-05T06:10:28.043457+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16991","citing_title":"Response-free item difficulty modelling for multiple-choice items with fine-tuned transformers: Component-wise representation and multi-task learning","ref_index":196,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TDGKLRGH5PPVBW4Y7ABBOSFRBX","json":"https://pith.science/pith/TDGKLRGH5PPVBW4Y7ABBOSFRBX.json","graph_json":"https://pith.science/api/pith-number/TDGKLRGH5PPVBW4Y7ABBOSFRBX/graph.json","events_json":"https://pith.science/api/pith-number/TDGKLRGH5PPVBW4Y7ABBOSFRBX/events.json","paper":"https://pith.science/paper/TDGKLRGH"},"agent_actions":{"view_html":"https://pith.science/pith/TDGKLRGH5PPVBW4Y7ABBOSFRBX","download_json":"https://pith.science/pith/TDGKLRGH5PPVBW4Y7ABBOSFRBX.json","view_paper":"https://pith.science/paper/TDGKLRGH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.05948&json=true","fetch_graph":"https://pith.science/api/pith-number/TDGKLRGH5PPVBW4Y7ABBOSFRBX/graph.json","fetch_events":"https://pith.science/api/pith-number/TDGKLRGH5PPVBW4Y7ABBOSFRBX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TDGKLRGH5PPVBW4Y7ABBOSFRBX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TDGKLRGH5PPVBW4Y7ABBOSFRBX/action/storage_attestation","attest_author":"https://pith.science/pith/TDGKLRGH5PPVBW4Y7ABBOSFRBX/action/author_attestation","sign_citation":"https://pith.science/pith/TDGKLRGH5PPVBW4Y7ABBOSFRBX/action/citation_signature","submit_replication":"https://pith.science/pith/TDGKLRGH5PPVBW4Y7ABBOSFRBX/action/replication_record"}},"created_at":"2026-07-05T06:10:28.043457+00:00","updated_at":"2026-07-05T06:10:28.043457+00:00"}