{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VP5ADBCFA7JUTQTBXYYRUPU4W4","short_pith_number":"pith:VP5ADBCF","schema_version":"1.0","canonical_sha256":"abfa01844507d349c261be311a3e9cb72d0259e53770048a7400301ccb3ca53e","source":{"kind":"arxiv","id":"2410.11840","version":2},"attestation_state":"computed","paper":{"title":"A Hitchhiker's Guide to Scaling Law Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jacob Andreas, Leshem Choshen, Yang Zhang","submitted_at":"2024-10-15T17:59:10Z","abstract_excerpt":"Scaling laws predict the loss of a target machine learning model by extrapolating from easier-to-train models with fewer parameters or smaller training sets. This provides an efficient way for practitioners and researchers alike to compare pretraining decisions involving optimizers, datasets, and model architectures. Despite the widespread use of scaling laws to model the dynamics of language model training, there has been little work on understanding how to best estimate and interpret them. We collect (and release) a large-scale dataset containing losses and downstream evaluations for 485 pre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.11840","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-15T17:59:10Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"0acb9cec70fa50223ac03934cbd0f630992e9f51f4a1cda7e537c2a388da6d64","abstract_canon_sha256":"8c4847ed99d5f6652a6e0b3ecde5199adc05c054903a75e87d120b73d5fa9edf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:25.108398Z","signature_b64":"OfUdjj2i6a8zpAOCoUB2zmTaQUTbwyBPGcu3SWEMf3HVP4IpxRdJ7esNKqYGCoGPTVm0e8nrncf3olShOn0DAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"abfa01844507d349c261be311a3e9cb72d0259e53770048a7400301ccb3ca53e","last_reissued_at":"2026-07-05T11:14:25.107965Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:25.107965Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Hitchhiker's Guide to Scaling Law Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jacob Andreas, Leshem Choshen, Yang Zhang","submitted_at":"2024-10-15T17:59:10Z","abstract_excerpt":"Scaling laws predict the loss of a target machine learning model by extrapolating from easier-to-train models with fewer parameters or smaller training sets. This provides an efficient way for practitioners and researchers alike to compare pretraining decisions involving optimizers, datasets, and model architectures. Despite the widespread use of scaling laws to model the dynamics of language model training, there has been little work on understanding how to best estimate and interpret them. We collect (and release) a large-scale dataset containing losses and downstream evaluations for 485 pre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.11840","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.11840/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.11840","created_at":"2026-07-05T11:14:25.108021+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.11840v2","created_at":"2026-07-05T11:14:25.108021+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.11840","created_at":"2026-07-05T11:14:25.108021+00:00"},{"alias_kind":"pith_short_12","alias_value":"VP5ADBCFA7JU","created_at":"2026-07-05T11:14:25.108021+00:00"},{"alias_kind":"pith_short_16","alias_value":"VP5ADBCFA7JUTQTB","created_at":"2026-07-05T11:14:25.108021+00:00"},{"alias_kind":"pith_short_8","alias_value":"VP5ADBCF","created_at":"2026-07-05T11:14:25.108021+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05029","citing_title":"Validity Threats for Foundation Model Research","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01045","citing_title":"Child-directed speech facilitates production, not comprehension, in BabyLMs","ref_index":183,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07546","citing_title":"On the Invariance and Generality of Neural Scaling Laws","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VP5ADBCFA7JUTQTBXYYRUPU4W4","json":"https://pith.science/pith/VP5ADBCFA7JUTQTBXYYRUPU4W4.json","graph_json":"https://pith.science/api/pith-number/VP5ADBCFA7JUTQTBXYYRUPU4W4/graph.json","events_json":"https://pith.science/api/pith-number/VP5ADBCFA7JUTQTBXYYRUPU4W4/events.json","paper":"https://pith.science/paper/VP5ADBCF"},"agent_actions":{"view_html":"https://pith.science/pith/VP5ADBCFA7JUTQTBXYYRUPU4W4","download_json":"https://pith.science/pith/VP5ADBCFA7JUTQTBXYYRUPU4W4.json","view_paper":"https://pith.science/paper/VP5ADBCF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.11840&json=true","fetch_graph":"https://pith.science/api/pith-number/VP5ADBCFA7JUTQTBXYYRUPU4W4/graph.json","fetch_events":"https://pith.science/api/pith-number/VP5ADBCFA7JUTQTBXYYRUPU4W4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VP5ADBCFA7JUTQTBXYYRUPU4W4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VP5ADBCFA7JUTQTBXYYRUPU4W4/action/storage_attestation","attest_author":"https://pith.science/pith/VP5ADBCFA7JUTQTBXYYRUPU4W4/action/author_attestation","sign_citation":"https://pith.science/pith/VP5ADBCFA7JUTQTBXYYRUPU4W4/action/citation_signature","submit_replication":"https://pith.science/pith/VP5ADBCFA7JUTQTBXYYRUPU4W4/action/replication_record"}},"created_at":"2026-07-05T11:14:25.108021+00:00","updated_at":"2026-07-05T11:14:25.108021+00:00"}