{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BYULAFPQFC3AHATEUOS64SPRZG","short_pith_number":"pith:BYULAFPQ","schema_version":"1.0","canonical_sha256":"0e28b015f028b6038264a3a5ee49f1c9b094f16fc4968f78a635db96e5a4768a","source":{"kind":"arxiv","id":"2310.15393","version":2},"attestation_state":"computed","paper":{"title":"DoGE: Domain Reweighting with Generalization Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Martin Jaggi, Matteo Pagliardini, Simin Fan","submitted_at":"2023-10-23T22:51:58Z","abstract_excerpt":"The coverage and composition of the pretraining data significantly impacts the generalization ability of Large Language Models (LLMs). Despite its importance, recent LLMs still rely on heuristics and trial and error to increase or reduce the influence of data-domains. We propose DOmain reweighting with Generalization Estimation (DoGE), which optimizes the probability of sampling from each domain (domain weights) in a principled way. Our approach is a two-stage process consisting of (i) training a proxy model to obtain domain weights using a bi-level optimization algorithm; (ii) training a larg"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.15393","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-23T22:51:58Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"4eae7797c9d5a1990ef8764c4794f18bb2a4bb63de2ead6d8e3b05775a55c971","abstract_canon_sha256":"8c022ca4ca1b3850b810fe3d27ec4a8973c82fbc1b5acbaca22a3d27fe6df968"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:41:28.047636Z","signature_b64":"LzxESxd9O6YmRY1oeqD8pKPuEkUteALJ01y9y8gcDr2P9igBq0/0uScG3cby7PPgpbgTmQN7Fgj52tubfu7NBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e28b015f028b6038264a3a5ee49f1c9b094f16fc4968f78a635db96e5a4768a","last_reissued_at":"2026-07-05T07:41:28.047209Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:41:28.047209Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DoGE: Domain Reweighting with Generalization Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Martin Jaggi, Matteo Pagliardini, Simin Fan","submitted_at":"2023-10-23T22:51:58Z","abstract_excerpt":"The coverage and composition of the pretraining data significantly impacts the generalization ability of Large Language Models (LLMs). Despite its importance, recent LLMs still rely on heuristics and trial and error to increase or reduce the influence of data-domains. We propose DOmain reweighting with Generalization Estimation (DoGE), which optimizes the probability of sampling from each domain (domain weights) in a principled way. Our approach is a two-stage process consisting of (i) training a proxy model to obtain domain weights using a bi-level optimization algorithm; (ii) training a larg"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.15393","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.15393/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.15393","created_at":"2026-07-05T07:41:28.047263+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.15393v2","created_at":"2026-07-05T07:41:28.047263+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.15393","created_at":"2026-07-05T07:41:28.047263+00:00"},{"alias_kind":"pith_short_12","alias_value":"BYULAFPQFC3A","created_at":"2026-07-05T07:41:28.047263+00:00"},{"alias_kind":"pith_short_16","alias_value":"BYULAFPQFC3AHATE","created_at":"2026-07-05T07:41:28.047263+00:00"},{"alias_kind":"pith_short_8","alias_value":"BYULAFPQ","created_at":"2026-07-05T07:41:28.047263+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01686","citing_title":"WARP: Weight-Space Analysis for Recovering Training Data Portfolios","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00571","citing_title":"On the Difficulty of Learning a Meta-network for Training Data Selection","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2502.00270","citing_title":"DUET: Optimizing Training Data Mixtures via Feedback from Unseen Evaluation Tasks","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11794","citing_title":"DataComp-LM: In search of the next generation of training sets for language models","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10288","citing_title":"BROS: Bias-Corrected Randomized Subspaces for Memory-Efficient Single-Loop Bilevel Optimization","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10288","citing_title":"BROS: Bias-Corrected Randomized Subspaces for Memory-Efficient Single-Loop Bilevel Optimization","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08366","citing_title":"Scaling-Aware Data Selection for End-to-End Autonomous Driving Systems","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08519","citing_title":"Cram Less to Fit More: Training Data Pruning Improves Memorization of Facts","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2505.09388","citing_title":"Qwen3 Technical Report","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BYULAFPQFC3AHATEUOS64SPRZG","json":"https://pith.science/pith/BYULAFPQFC3AHATEUOS64SPRZG.json","graph_json":"https://pith.science/api/pith-number/BYULAFPQFC3AHATEUOS64SPRZG/graph.json","events_json":"https://pith.science/api/pith-number/BYULAFPQFC3AHATEUOS64SPRZG/events.json","paper":"https://pith.science/paper/BYULAFPQ"},"agent_actions":{"view_html":"https://pith.science/pith/BYULAFPQFC3AHATEUOS64SPRZG","download_json":"https://pith.science/pith/BYULAFPQFC3AHATEUOS64SPRZG.json","view_paper":"https://pith.science/paper/BYULAFPQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.15393&json=true","fetch_graph":"https://pith.science/api/pith-number/BYULAFPQFC3AHATEUOS64SPRZG/graph.json","fetch_events":"https://pith.science/api/pith-number/BYULAFPQFC3AHATEUOS64SPRZG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BYULAFPQFC3AHATEUOS64SPRZG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BYULAFPQFC3AHATEUOS64SPRZG/action/storage_attestation","attest_author":"https://pith.science/pith/BYULAFPQFC3AHATEUOS64SPRZG/action/author_attestation","sign_citation":"https://pith.science/pith/BYULAFPQFC3AHATEUOS64SPRZG/action/citation_signature","submit_replication":"https://pith.science/pith/BYULAFPQFC3AHATEUOS64SPRZG/action/replication_record"}},"created_at":"2026-07-05T07:41:28.047263+00:00","updated_at":"2026-07-05T07:41:28.047263+00:00"}