{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:LDW4QRJII4SHMZHIRVYP5L4Y6Y","short_pith_number":"pith:LDW4QRJI","schema_version":"1.0","canonical_sha256":"58edc8452847247664e88d70feaf98f618c0a22a816797518ff1ccd0c681bf03","source":{"kind":"arxiv","id":"1906.01827","version":3},"attestation_state":"computed","paper":{"title":"Coresets for Data-efficient Training of Machine Learning Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Baharan Mirzasoleiman, Jeff Bilmes, Jure Leskovec","submitted_at":"2019-06-05T05:10:37Z","abstract_excerpt":"Incremental gradient (IG) methods, such as stochastic gradient descent and its variants are commonly used for large scale optimization in machine learning. Despite the sustained effort to make IG methods more data-efficient, it remains an open question how to select a training data subset that can theoretically and practically perform on par with the full dataset. Here we develop CRAIG, a method to select a weighted subset (or coreset) of training data that closely estimates the full gradient by maximizing a submodular function. We prove that applying IG to this subset is guaranteed to converg"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1906.01827","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-06-05T05:10:37Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"cda9e8dba12c056c1fdaddc80e13e80ea740e84d273139720d4ef6ea1cfad212","abstract_canon_sha256":"101e553783c31181164c51363e6130bbcbf9f382fae065c6259b74879ecc901d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:51:58.805934Z","signature_b64":"UeWJaQ/l9KzUOPCF2zFUVIGUq/X/k2fwFJtJnpBJcuCcLwJ9EF3mBzfWOBcjIoci3vNzx2F0T/rxH+9swSheAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"58edc8452847247664e88d70feaf98f618c0a22a816797518ff1ccd0c681bf03","last_reissued_at":"2026-07-05T01:51:58.805433Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:51:58.805433Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Coresets for Data-efficient Training of Machine Learning Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Baharan Mirzasoleiman, Jeff Bilmes, Jure Leskovec","submitted_at":"2019-06-05T05:10:37Z","abstract_excerpt":"Incremental gradient (IG) methods, such as stochastic gradient descent and its variants are commonly used for large scale optimization in machine learning. Despite the sustained effort to make IG methods more data-efficient, it remains an open question how to select a training data subset that can theoretically and practically perform on par with the full dataset. Here we develop CRAIG, a method to select a weighted subset (or coreset) of training data that closely estimates the full gradient by maximizing a submodular function. We prove that applying IG to this subset is guaranteed to converg"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1906.01827","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1906.01827/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1906.01827","created_at":"2026-07-05T01:51:58.805489+00:00"},{"alias_kind":"arxiv_version","alias_value":"1906.01827v3","created_at":"2026-07-05T01:51:58.805489+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1906.01827","created_at":"2026-07-05T01:51:58.805489+00:00"},{"alias_kind":"pith_short_12","alias_value":"LDW4QRJII4SH","created_at":"2026-07-05T01:51:58.805489+00:00"},{"alias_kind":"pith_short_16","alias_value":"LDW4QRJII4SHMZHI","created_at":"2026-07-05T01:51:58.805489+00:00"},{"alias_kind":"pith_short_8","alias_value":"LDW4QRJI","created_at":"2026-07-05T01:51:58.805489+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.00270","citing_title":"DUET: Optimizing Training Data Mixtures via Feedback from Unseen Evaluation Tasks","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2407.12772","citing_title":"LMMs-Eval: Reality Check on the Evaluation of Large Multimodal Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09404","citing_title":"Let the Target Select for Itself: Data Selection via Target-Aligned Paths","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11810","citing_title":"GRACE: A Dynamic Coreset Selection Framework for Large Language Model Optimization","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02647","citing_title":"ContextualJailbreak: Evolutionary Red-Teaming via Simulated Conversational Priming","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LDW4QRJII4SHMZHIRVYP5L4Y6Y","json":"https://pith.science/pith/LDW4QRJII4SHMZHIRVYP5L4Y6Y.json","graph_json":"https://pith.science/api/pith-number/LDW4QRJII4SHMZHIRVYP5L4Y6Y/graph.json","events_json":"https://pith.science/api/pith-number/LDW4QRJII4SHMZHIRVYP5L4Y6Y/events.json","paper":"https://pith.science/paper/LDW4QRJI"},"agent_actions":{"view_html":"https://pith.science/pith/LDW4QRJII4SHMZHIRVYP5L4Y6Y","download_json":"https://pith.science/pith/LDW4QRJII4SHMZHIRVYP5L4Y6Y.json","view_paper":"https://pith.science/paper/LDW4QRJI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1906.01827&json=true","fetch_graph":"https://pith.science/api/pith-number/LDW4QRJII4SHMZHIRVYP5L4Y6Y/graph.json","fetch_events":"https://pith.science/api/pith-number/LDW4QRJII4SHMZHIRVYP5L4Y6Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LDW4QRJII4SHMZHIRVYP5L4Y6Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LDW4QRJII4SHMZHIRVYP5L4Y6Y/action/storage_attestation","attest_author":"https://pith.science/pith/LDW4QRJII4SHMZHIRVYP5L4Y6Y/action/author_attestation","sign_citation":"https://pith.science/pith/LDW4QRJII4SHMZHIRVYP5L4Y6Y/action/citation_signature","submit_replication":"https://pith.science/pith/LDW4QRJII4SHMZHIRVYP5L4Y6Y/action/replication_record"}},"created_at":"2026-07-05T01:51:58.805489+00:00","updated_at":"2026-07-05T01:51:58.805489+00:00"}