{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YO2OEPMSLTNSXYCX3KIVDK7SFX","short_pith_number":"pith:YO2OEPMS","schema_version":"1.0","canonical_sha256":"c3b4e23d925cdb2be057da9151abf22dc6d8d30a2d93d49dbbe431391e98966e","source":{"kind":"arxiv","id":"2402.03985","version":3},"attestation_state":"computed","paper":{"title":"A Bias-Variance Decomposition for Ensembles over Multiple Synthetic Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Antti Honkela, Ossi R\\\"ais\\\"a","submitted_at":"2024-02-06T13:20:46Z","abstract_excerpt":"Recent studies have highlighted the benefits of generating multiple synthetic datasets for supervised learning, from increased accuracy to more effective model selection and uncertainty estimation. These benefits have clear empirical support, but the theoretical understanding of them is currently very light. We seek to increase the theoretical understanding by deriving bias-variance decompositions for several settings of using multiple synthetic datasets, including differentially private synthetic data. Our theory yields a simple rule of thumb to select the appropriate number of synthetic data"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.03985","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-06T13:20:46Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"7191d39d489c1acad71fcf67347b681312c51df22b6425d5a48e161f3c70abde","abstract_canon_sha256":"8c42902300a2dc6484346be2e8f868a6e5838c9206a4a0281c2d486113544947"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:53:44.127076Z","signature_b64":"8kEqaF80iHPpAJmDFbqZRYMSQ8Ylk9jjBf9Mx2fZXu8SKb8De8F0zp+RiFHGH934fAQzDdqOdlZce3iUAhWzCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c3b4e23d925cdb2be057da9151abf22dc6d8d30a2d93d49dbbe431391e98966e","last_reissued_at":"2026-07-05T10:53:44.126677Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:53:44.126677Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Bias-Variance Decomposition for Ensembles over Multiple Synthetic Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Antti Honkela, Ossi R\\\"ais\\\"a","submitted_at":"2024-02-06T13:20:46Z","abstract_excerpt":"Recent studies have highlighted the benefits of generating multiple synthetic datasets for supervised learning, from increased accuracy to more effective model selection and uncertainty estimation. These benefits have clear empirical support, but the theoretical understanding of them is currently very light. We seek to increase the theoretical understanding by deriving bias-variance decompositions for several settings of using multiple synthetic datasets, including differentially private synthetic data. Our theory yields a simple rule of thumb to select the appropriate number of synthetic data"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.03985","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.03985/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.03985","created_at":"2026-07-05T10:53:44.126727+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.03985v3","created_at":"2026-07-05T10:53:44.126727+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.03985","created_at":"2026-07-05T10:53:44.126727+00:00"},{"alias_kind":"pith_short_12","alias_value":"YO2OEPMSLTNS","created_at":"2026-07-05T10:53:44.126727+00:00"},{"alias_kind":"pith_short_16","alias_value":"YO2OEPMSLTNSXYCX","created_at":"2026-07-05T10:53:44.126727+00:00"},{"alias_kind":"pith_short_8","alias_value":"YO2OEPMS","created_at":"2026-07-05T10:53:44.126727+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.24190","citing_title":"Provably Improving Generalization of Few-Shot Models with Synthetic Data","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YO2OEPMSLTNSXYCX3KIVDK7SFX","json":"https://pith.science/pith/YO2OEPMSLTNSXYCX3KIVDK7SFX.json","graph_json":"https://pith.science/api/pith-number/YO2OEPMSLTNSXYCX3KIVDK7SFX/graph.json","events_json":"https://pith.science/api/pith-number/YO2OEPMSLTNSXYCX3KIVDK7SFX/events.json","paper":"https://pith.science/paper/YO2OEPMS"},"agent_actions":{"view_html":"https://pith.science/pith/YO2OEPMSLTNSXYCX3KIVDK7SFX","download_json":"https://pith.science/pith/YO2OEPMSLTNSXYCX3KIVDK7SFX.json","view_paper":"https://pith.science/paper/YO2OEPMS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.03985&json=true","fetch_graph":"https://pith.science/api/pith-number/YO2OEPMSLTNSXYCX3KIVDK7SFX/graph.json","fetch_events":"https://pith.science/api/pith-number/YO2OEPMSLTNSXYCX3KIVDK7SFX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YO2OEPMSLTNSXYCX3KIVDK7SFX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YO2OEPMSLTNSXYCX3KIVDK7SFX/action/storage_attestation","attest_author":"https://pith.science/pith/YO2OEPMSLTNSXYCX3KIVDK7SFX/action/author_attestation","sign_citation":"https://pith.science/pith/YO2OEPMSLTNSXYCX3KIVDK7SFX/action/citation_signature","submit_replication":"https://pith.science/pith/YO2OEPMSLTNSXYCX3KIVDK7SFX/action/replication_record"}},"created_at":"2026-07-05T10:53:44.126727+00:00","updated_at":"2026-07-05T10:53:44.126727+00:00"}