{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5DM6XWEQFQEUFE65KSXDPQ5PBM","short_pith_number":"pith:5DM6XWEQ","schema_version":"1.0","canonical_sha256":"e8d9ebd8902c094293dd54ae37c3af0b3afc7eba2cc3cd6942128895fd396fb7","source":{"kind":"arxiv","id":"2411.12925","version":1},"attestation_state":"computed","paper":{"title":"Loss-to-Loss Prediction: Scaling Laws for All Datasets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"David Brandfonbrener, Eran Malach, Nikhil Anand, Nikhil Vyas, Sham Kakade","submitted_at":"2024-11-19T23:23:16Z","abstract_excerpt":"While scaling laws provide a reliable methodology for predicting train loss across compute scales for a single data distribution, less is known about how these predictions should change as we change the distribution. In this paper, we derive a strategy for predicting one loss from another and apply it to predict across different pre-training datasets and from pre-training data to downstream task data. Our predictions extrapolate well even at 20x the largest FLOP budget used to fit the curves. More precisely, we find that there are simple shifted power law relationships between (1) the train lo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.12925","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-11-19T23:23:16Z","cross_cats_sorted":["cs.AI","cs.CL","stat.ML"],"title_canon_sha256":"9ab8921581628dd2077d398afc280ee6d1cf2e69425379c4d43b3bf558849040","abstract_canon_sha256":"f86912ecfae38e397c94bae6a39d73c80003026e63d6236c127b96d340dec07b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:02.384742Z","signature_b64":"cxM7dgeufOFku2kzn6R9s99ZXGCOfRheKbho/IDZiuGqaPlWceNJlLO8DhXcedowrzF4+Dy9gLHBjuxOcqqsAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e8d9ebd8902c094293dd54ae37c3af0b3afc7eba2cc3cd6942128895fd396fb7","last_reissued_at":"2026-07-05T09:38:02.384151Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:02.384151Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Loss-to-Loss Prediction: Scaling Laws for All Datasets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"David Brandfonbrener, Eran Malach, Nikhil Anand, Nikhil Vyas, Sham Kakade","submitted_at":"2024-11-19T23:23:16Z","abstract_excerpt":"While scaling laws provide a reliable methodology for predicting train loss across compute scales for a single data distribution, less is known about how these predictions should change as we change the distribution. In this paper, we derive a strategy for predicting one loss from another and apply it to predict across different pre-training datasets and from pre-training data to downstream task data. Our predictions extrapolate well even at 20x the largest FLOP budget used to fit the curves. More precisely, we find that there are simple shifted power law relationships between (1) the train lo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.12925","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.12925/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.12925","created_at":"2026-07-05T09:38:02.384211+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.12925v1","created_at":"2026-07-05T09:38:02.384211+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.12925","created_at":"2026-07-05T09:38:02.384211+00:00"},{"alias_kind":"pith_short_12","alias_value":"5DM6XWEQFQEU","created_at":"2026-07-05T09:38:02.384211+00:00"},{"alias_kind":"pith_short_16","alias_value":"5DM6XWEQFQEUFE65","created_at":"2026-07-05T09:38:02.384211+00:00"},{"alias_kind":"pith_short_8","alias_value":"5DM6XWEQ","created_at":"2026-07-05T09:38:02.384211+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.12120","citing_title":"LLMs on the Line: Data Determines Loss-to-Loss Scaling Laws","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07546","citing_title":"On the Invariance and Generality of Neural Scaling Laws","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14815","citing_title":"Domain Fine-Tuning FinBERT on Finnish Histopathological Reports: Train-Time Signals and Downstream Correlations","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5DM6XWEQFQEUFE65KSXDPQ5PBM","json":"https://pith.science/pith/5DM6XWEQFQEUFE65KSXDPQ5PBM.json","graph_json":"https://pith.science/api/pith-number/5DM6XWEQFQEUFE65KSXDPQ5PBM/graph.json","events_json":"https://pith.science/api/pith-number/5DM6XWEQFQEUFE65KSXDPQ5PBM/events.json","paper":"https://pith.science/paper/5DM6XWEQ"},"agent_actions":{"view_html":"https://pith.science/pith/5DM6XWEQFQEUFE65KSXDPQ5PBM","download_json":"https://pith.science/pith/5DM6XWEQFQEUFE65KSXDPQ5PBM.json","view_paper":"https://pith.science/paper/5DM6XWEQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.12925&json=true","fetch_graph":"https://pith.science/api/pith-number/5DM6XWEQFQEUFE65KSXDPQ5PBM/graph.json","fetch_events":"https://pith.science/api/pith-number/5DM6XWEQFQEUFE65KSXDPQ5PBM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5DM6XWEQFQEUFE65KSXDPQ5PBM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5DM6XWEQFQEUFE65KSXDPQ5PBM/action/storage_attestation","attest_author":"https://pith.science/pith/5DM6XWEQFQEUFE65KSXDPQ5PBM/action/author_attestation","sign_citation":"https://pith.science/pith/5DM6XWEQFQEUFE65KSXDPQ5PBM/action/citation_signature","submit_replication":"https://pith.science/pith/5DM6XWEQFQEUFE65KSXDPQ5PBM/action/replication_record"}},"created_at":"2026-07-05T09:38:02.384211+00:00","updated_at":"2026-07-05T09:38:02.384211+00:00"}