{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JROKZP6PUEWH2QCKINQ6AEMN6G","short_pith_number":"pith:JROKZP6P","schema_version":"1.0","canonical_sha256":"4c5cacbfcfa12c7d404a4361e0118df184127be606ad4d67cefabfbe3c98b698","source":{"kind":"arxiv","id":"2412.04403","version":2},"attestation_state":"computed","paper":{"title":"Establishing Task Scaling Laws via Compute-Efficient Model Ladders","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Akshita Bhagia, Alexander Wettig, Ananya Harsh Jha, David Heineman, Dirk Groeneveld, Hannaneh Hajishirzi, Jesse Dodge, Jiacheng Liu, Luca Soldaini, Noah A. Smith, Oyvind Tafjord, Pang Wei Koh","submitted_at":"2024-12-05T18:21:49Z","abstract_excerpt":"We develop task scaling laws and model ladders to predict the individual task performance of pretrained language models (LMs) in the overtrained setting. Standard power laws for language modeling loss cannot accurately model task performance. Therefore, we leverage a two-step prediction approach: (1) use model and data size to predict an intermediate loss, then (2) use it to predict task performance. We train a set of small-scale \"ladder\" models, collect data points to fit the parameterized functions of the two prediction steps, and make predictions for two target models: a 7B model trained to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.04403","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-05T18:21:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"546887215a99731a8cafe0f505a10307f5fc118e155d2e1ae8c590f0a468125d","abstract_canon_sha256":"e887e7bfa1bd12665cb56f1ea408d38262b988eebd2d6c5395cc2a7ce0a6716c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:57:33.196085Z","signature_b64":"pDWiJvTTmvBKWvpgNK/7WWIBo5HCs3Jt6trzuTRtQCss9ZoW5fnVSPsBnFO4tQWxpWNajKNbr8/iAfY2uLAtAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c5cacbfcfa12c7d404a4361e0118df184127be606ad4d67cefabfbe3c98b698","last_reissued_at":"2026-07-05T11:57:33.195617Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:57:33.195617Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Establishing Task Scaling Laws via Compute-Efficient Model Ladders","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Akshita Bhagia, Alexander Wettig, Ananya Harsh Jha, David Heineman, Dirk Groeneveld, Hannaneh Hajishirzi, Jesse Dodge, Jiacheng Liu, Luca Soldaini, Noah A. Smith, Oyvind Tafjord, Pang Wei Koh","submitted_at":"2024-12-05T18:21:49Z","abstract_excerpt":"We develop task scaling laws and model ladders to predict the individual task performance of pretrained language models (LMs) in the overtrained setting. Standard power laws for language modeling loss cannot accurately model task performance. Therefore, we leverage a two-step prediction approach: (1) use model and data size to predict an intermediate loss, then (2) use it to predict task performance. We train a set of small-scale \"ladder\" models, collect data points to fit the parameterized functions of the two prediction steps, and make predictions for two target models: a 7B model trained to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.04403","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.04403/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.04403","created_at":"2026-07-05T11:57:33.195673+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.04403v2","created_at":"2026-07-05T11:57:33.195673+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.04403","created_at":"2026-07-05T11:57:33.195673+00:00"},{"alias_kind":"pith_short_12","alias_value":"JROKZP6PUEWH","created_at":"2026-07-05T11:57:33.195673+00:00"},{"alias_kind":"pith_short_16","alias_value":"JROKZP6PUEWH2QCK","created_at":"2026-07-05T11:57:33.195673+00:00"},{"alias_kind":"pith_short_8","alias_value":"JROKZP6P","created_at":"2026-07-05T11:57:33.195673+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07616","citing_title":"Item Response Scaling Laws: A Measurement Theory Approach for Efficient and Generalizable Neural Scaling Estimation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08022","citing_title":"Capacity-Aware Mixture Law Enables Efficient LLM Data Optimization","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24416","citing_title":"Scaling Properties of Continuous Diffusion Spoken Language Models","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21215","citing_title":"The Recurrent Transformer: Greater Effective Depth and Efficient Decoding","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JROKZP6PUEWH2QCKINQ6AEMN6G","json":"https://pith.science/pith/JROKZP6PUEWH2QCKINQ6AEMN6G.json","graph_json":"https://pith.science/api/pith-number/JROKZP6PUEWH2QCKINQ6AEMN6G/graph.json","events_json":"https://pith.science/api/pith-number/JROKZP6PUEWH2QCKINQ6AEMN6G/events.json","paper":"https://pith.science/paper/JROKZP6P"},"agent_actions":{"view_html":"https://pith.science/pith/JROKZP6PUEWH2QCKINQ6AEMN6G","download_json":"https://pith.science/pith/JROKZP6PUEWH2QCKINQ6AEMN6G.json","view_paper":"https://pith.science/paper/JROKZP6P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.04403&json=true","fetch_graph":"https://pith.science/api/pith-number/JROKZP6PUEWH2QCKINQ6AEMN6G/graph.json","fetch_events":"https://pith.science/api/pith-number/JROKZP6PUEWH2QCKINQ6AEMN6G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JROKZP6PUEWH2QCKINQ6AEMN6G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JROKZP6PUEWH2QCKINQ6AEMN6G/action/storage_attestation","attest_author":"https://pith.science/pith/JROKZP6PUEWH2QCKINQ6AEMN6G/action/author_attestation","sign_citation":"https://pith.science/pith/JROKZP6PUEWH2QCKINQ6AEMN6G/action/citation_signature","submit_replication":"https://pith.science/pith/JROKZP6PUEWH2QCKINQ6AEMN6G/action/replication_record"}},"created_at":"2026-07-05T11:57:33.195673+00:00","updated_at":"2026-07-05T11:57:33.195673+00:00"}