{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:N6ZOVFZQO6NRLBBRKHCMZDUXIA","short_pith_number":"pith:N6ZOVFZQ","schema_version":"1.0","canonical_sha256":"6fb2ea9730779b15843151c4cc8e974010d5855687006e58b3d301d4ac210cf7","source":{"kind":"arxiv","id":"2506.11271","version":1},"attestation_state":"computed","paper":{"title":"Collaborative Prediction: To Join or To Disjoin Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Guanting Chen, Kyung Rok Kim, Xiaocheng Li, Yansong Wang","submitted_at":"2025-06-12T20:25:07Z","abstract_excerpt":"With the recent rise of generative Artificial Intelligence (AI), the need of selecting high-quality dataset to improve machine learning models has garnered increasing attention. However, some part of this topic remains underexplored, even for simple prediction models. In this work, we study the problem of developing practical algorithms that select appropriate dataset to minimize population loss of our prediction model with high probability. Broadly speaking, we investigate when datasets from different sources can be effectively merged to enhance the predictive model's performance, and propose"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.11271","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ML","submitted_at":"2025-06-12T20:25:07Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a223f3de4172bc5634cf30c8ecd9dd07d1a6800f569b46e02a59a2c0cb0792f5","abstract_canon_sha256":"b4917af71fbb722dd2d21b224234d6a3438e01123e0fde219a53b6ad065cd336"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:21:01.923807Z","signature_b64":"b6De2GyRsIpfp2VpjNVbUVUpw6Zx1H1ZqCpzQFnjDEp54hz5XcR4RJEjd/CIF4P78ylOqdMQstUDJwl1MCtXCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6fb2ea9730779b15843151c4cc8e974010d5855687006e58b3d301d4ac210cf7","last_reissued_at":"2026-07-05T11:21:01.923345Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:21:01.923345Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Collaborative Prediction: To Join or To Disjoin Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Guanting Chen, Kyung Rok Kim, Xiaocheng Li, Yansong Wang","submitted_at":"2025-06-12T20:25:07Z","abstract_excerpt":"With the recent rise of generative Artificial Intelligence (AI), the need of selecting high-quality dataset to improve machine learning models has garnered increasing attention. However, some part of this topic remains underexplored, even for simple prediction models. In this work, we study the problem of developing practical algorithms that select appropriate dataset to minimize population loss of our prediction model with high probability. Broadly speaking, we investigate when datasets from different sources can be effectively merged to enhance the predictive model's performance, and propose"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.11271","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.11271/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.11271","created_at":"2026-07-05T11:21:01.923404+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.11271v1","created_at":"2026-07-05T11:21:01.923404+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.11271","created_at":"2026-07-05T11:21:01.923404+00:00"},{"alias_kind":"pith_short_12","alias_value":"N6ZOVFZQO6NR","created_at":"2026-07-05T11:21:01.923404+00:00"},{"alias_kind":"pith_short_16","alias_value":"N6ZOVFZQO6NRLBBR","created_at":"2026-07-05T11:21:01.923404+00:00"},{"alias_kind":"pith_short_8","alias_value":"N6ZOVFZQ","created_at":"2026-07-05T11:21:01.923404+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N6ZOVFZQO6NRLBBRKHCMZDUXIA","json":"https://pith.science/pith/N6ZOVFZQO6NRLBBRKHCMZDUXIA.json","graph_json":"https://pith.science/api/pith-number/N6ZOVFZQO6NRLBBRKHCMZDUXIA/graph.json","events_json":"https://pith.science/api/pith-number/N6ZOVFZQO6NRLBBRKHCMZDUXIA/events.json","paper":"https://pith.science/paper/N6ZOVFZQ"},"agent_actions":{"view_html":"https://pith.science/pith/N6ZOVFZQO6NRLBBRKHCMZDUXIA","download_json":"https://pith.science/pith/N6ZOVFZQO6NRLBBRKHCMZDUXIA.json","view_paper":"https://pith.science/paper/N6ZOVFZQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.11271&json=true","fetch_graph":"https://pith.science/api/pith-number/N6ZOVFZQO6NRLBBRKHCMZDUXIA/graph.json","fetch_events":"https://pith.science/api/pith-number/N6ZOVFZQO6NRLBBRKHCMZDUXIA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N6ZOVFZQO6NRLBBRKHCMZDUXIA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N6ZOVFZQO6NRLBBRKHCMZDUXIA/action/storage_attestation","attest_author":"https://pith.science/pith/N6ZOVFZQO6NRLBBRKHCMZDUXIA/action/author_attestation","sign_citation":"https://pith.science/pith/N6ZOVFZQO6NRLBBRKHCMZDUXIA/action/citation_signature","submit_replication":"https://pith.science/pith/N6ZOVFZQO6NRLBBRKHCMZDUXIA/action/replication_record"}},"created_at":"2026-07-05T11:21:01.923404+00:00","updated_at":"2026-07-05T11:21:01.923404+00:00"}