{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IIUJ23DHZQC2Q4SA72Y7UPB24L","short_pith_number":"pith:IIUJ23DH","schema_version":"1.0","canonical_sha256":"42289d6c67cc05a87240feb1fa3c3ae2ebd9a8a097a6d0677f1582889066839a","source":{"kind":"arxiv","id":"2502.15588","version":1},"attestation_state":"computed","paper":{"title":"Improving the Scaling Laws of Synthetic Data with Deliberate Practice","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Adriana Romero-Soriano, Elvis Dohmatob, Florian Bordes, Jakob Verbeek, Melissa Hall, Michal Drozdzal, Mohammad Pezeshki, Pietro Astolfi, Reyhane Askari-Hemmat","submitted_at":"2025-02-21T16:56:15Z","abstract_excerpt":"Inspired by the principle of deliberate practice in human learning, we propose Deliberate Practice for Synthetic Data Generation (DP), a novel framework that improves sample efficiency through dynamic synthetic data generation. Prior work has shown that scaling synthetic data is inherently challenging, as naively adding new data leads to diminishing returns. To address this, pruning has been identified as a key mechanism for improving scaling, enabling models to focus on the most informative synthetic samples. Rather than generating a large dataset and pruning it afterward, DP efficiently appr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.15588","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-21T16:56:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8780f8718c084967779c8b89b4e2b80d228b741974a847286c1573538d7e4b44","abstract_canon_sha256":"0e76b11036b682c5d3a6601368fb003e766d7ac34d6aa65e0bd6b4f079c4c214"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:03.965181Z","signature_b64":"oLCJKF7bDKZSDz6lo9TCYgVPlbYP/EnJZwMUhUUAgDWDcsaZ0XP4u9UvT4VHbG/ZMOG3VjAt83Mdrn4xDv2sCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"42289d6c67cc05a87240feb1fa3c3ae2ebd9a8a097a6d0677f1582889066839a","last_reissued_at":"2026-07-05T10:18:03.964748Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:03.964748Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving the Scaling Laws of Synthetic Data with Deliberate Practice","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Adriana Romero-Soriano, Elvis Dohmatob, Florian Bordes, Jakob Verbeek, Melissa Hall, Michal Drozdzal, Mohammad Pezeshki, Pietro Astolfi, Reyhane Askari-Hemmat","submitted_at":"2025-02-21T16:56:15Z","abstract_excerpt":"Inspired by the principle of deliberate practice in human learning, we propose Deliberate Practice for Synthetic Data Generation (DP), a novel framework that improves sample efficiency through dynamic synthetic data generation. Prior work has shown that scaling synthetic data is inherently challenging, as naively adding new data leads to diminishing returns. To address this, pruning has been identified as a key mechanism for improving scaling, enabling models to focus on the most informative synthetic samples. Rather than generating a large dataset and pruning it afterward, DP efficiently appr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.15588","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.15588/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.15588","created_at":"2026-07-05T10:18:03.964807+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.15588v1","created_at":"2026-07-05T10:18:03.964807+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.15588","created_at":"2026-07-05T10:18:03.964807+00:00"},{"alias_kind":"pith_short_12","alias_value":"IIUJ23DHZQC2","created_at":"2026-07-05T10:18:03.964807+00:00"},{"alias_kind":"pith_short_16","alias_value":"IIUJ23DHZQC2Q4SA","created_at":"2026-07-05T10:18:03.964807+00:00"},{"alias_kind":"pith_short_8","alias_value":"IIUJ23DH","created_at":"2026-07-05T10:18:03.964807+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.11231","citing_title":"LiBaGS: Lightweight Boundary Gap Synthesis for Targeted Synthetic Data Selection","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11231","citing_title":"LiBaGS: Lightweight Boundary Gap Synthesis for Targeted Synthetic Data Selection","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IIUJ23DHZQC2Q4SA72Y7UPB24L","json":"https://pith.science/pith/IIUJ23DHZQC2Q4SA72Y7UPB24L.json","graph_json":"https://pith.science/api/pith-number/IIUJ23DHZQC2Q4SA72Y7UPB24L/graph.json","events_json":"https://pith.science/api/pith-number/IIUJ23DHZQC2Q4SA72Y7UPB24L/events.json","paper":"https://pith.science/paper/IIUJ23DH"},"agent_actions":{"view_html":"https://pith.science/pith/IIUJ23DHZQC2Q4SA72Y7UPB24L","download_json":"https://pith.science/pith/IIUJ23DHZQC2Q4SA72Y7UPB24L.json","view_paper":"https://pith.science/paper/IIUJ23DH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.15588&json=true","fetch_graph":"https://pith.science/api/pith-number/IIUJ23DHZQC2Q4SA72Y7UPB24L/graph.json","fetch_events":"https://pith.science/api/pith-number/IIUJ23DHZQC2Q4SA72Y7UPB24L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IIUJ23DHZQC2Q4SA72Y7UPB24L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IIUJ23DHZQC2Q4SA72Y7UPB24L/action/storage_attestation","attest_author":"https://pith.science/pith/IIUJ23DHZQC2Q4SA72Y7UPB24L/action/author_attestation","sign_citation":"https://pith.science/pith/IIUJ23DHZQC2Q4SA72Y7UPB24L/action/citation_signature","submit_replication":"https://pith.science/pith/IIUJ23DHZQC2Q4SA72Y7UPB24L/action/replication_record"}},"created_at":"2026-07-05T10:18:03.964807+00:00","updated_at":"2026-07-05T10:18:03.964807+00:00"}