{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FRQASMNILNRZUZ7M6WJM2O3TI4","short_pith_number":"pith:FRQASMNI","schema_version":"1.0","canonical_sha256":"2c600931a85b639a67ecf592cd3b73472f8205bff215b90de9c6f96c64731f81","source":{"kind":"arxiv","id":"2501.01956","version":3},"attestation_state":"computed","paper":{"title":"Metadata Conditioning Accelerates Language Model Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexander Wettig, Danqi Chen, Luxi He, Sadhika Malladi, Tianyu Gao, Yihe Dong","submitted_at":"2025-01-03T18:59:23Z","abstract_excerpt":"The vast diversity of styles, domains, and quality levels present in language model pre-training corpora is essential in developing general model capabilities, but efficiently learning and deploying the correct behaviors exemplified in each of these heterogeneous data sources is challenging. To address this, we propose a new method, termed Metadata Conditioning then Cooldown (MeCo), to incorporate additional learning cues during pre-training. MeCo first provides metadata (e.g., URLs like www$.$wikipedia$.$org) alongside the text during training and later uses a cooldown phase with only the sta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.01956","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-01-03T18:59:23Z","cross_cats_sorted":[],"title_canon_sha256":"fbf1efc001df94cf10fcb20dc14b4392e4ea9d47ca1a4ebc7957fa8382b39559","abstract_canon_sha256":"e8c532f87e00d12bb20a9125642649ac87cb111f131478f1b48941393a4ddf00"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:13.379297Z","signature_b64":"ANtf0vy0869dfg8FO77rLLZT7itq6V0Ddd3Q2Swtb1cRnFLs8G6+H9A7KLsp0rsDgsHzQt4BtrXKzMlas4VHBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c600931a85b639a67ecf592cd3b73472f8205bff215b90de9c6f96c64731f81","last_reissued_at":"2026-07-05T11:28:13.378802Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:13.378802Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Metadata Conditioning Accelerates Language Model Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexander Wettig, Danqi Chen, Luxi He, Sadhika Malladi, Tianyu Gao, Yihe Dong","submitted_at":"2025-01-03T18:59:23Z","abstract_excerpt":"The vast diversity of styles, domains, and quality levels present in language model pre-training corpora is essential in developing general model capabilities, but efficiently learning and deploying the correct behaviors exemplified in each of these heterogeneous data sources is challenging. To address this, we propose a new method, termed Metadata Conditioning then Cooldown (MeCo), to incorporate additional learning cues during pre-training. MeCo first provides metadata (e.g., URLs like www$.$wikipedia$.$org) alongside the text during training and later uses a cooldown phase with only the sta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.01956","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.01956/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.01956","created_at":"2026-07-05T11:28:13.378864+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.01956v3","created_at":"2026-07-05T11:28:13.378864+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.01956","created_at":"2026-07-05T11:28:13.378864+00:00"},{"alias_kind":"pith_short_12","alias_value":"FRQASMNILNRZ","created_at":"2026-07-05T11:28:13.378864+00:00"},{"alias_kind":"pith_short_16","alias_value":"FRQASMNILNRZUZ7M","created_at":"2026-07-05T11:28:13.378864+00:00"},{"alias_kind":"pith_short_8","alias_value":"FRQASMNI","created_at":"2026-07-05T11:28:13.378864+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.15523","citing_title":"Self-Prompting Diffusion Transformer for Open-Vocabulary Scene Text Editing via In-Context Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15523","citing_title":"Self-Prompting Diffusion Transformer for Open-Vocabulary Scene Text Editing via In-Context Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21613","citing_title":"Beyond URLs: Metadata Diversity and Position for Efficient LLM Pretraining","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FRQASMNILNRZUZ7M6WJM2O3TI4","json":"https://pith.science/pith/FRQASMNILNRZUZ7M6WJM2O3TI4.json","graph_json":"https://pith.science/api/pith-number/FRQASMNILNRZUZ7M6WJM2O3TI4/graph.json","events_json":"https://pith.science/api/pith-number/FRQASMNILNRZUZ7M6WJM2O3TI4/events.json","paper":"https://pith.science/paper/FRQASMNI"},"agent_actions":{"view_html":"https://pith.science/pith/FRQASMNILNRZUZ7M6WJM2O3TI4","download_json":"https://pith.science/pith/FRQASMNILNRZUZ7M6WJM2O3TI4.json","view_paper":"https://pith.science/paper/FRQASMNI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.01956&json=true","fetch_graph":"https://pith.science/api/pith-number/FRQASMNILNRZUZ7M6WJM2O3TI4/graph.json","fetch_events":"https://pith.science/api/pith-number/FRQASMNILNRZUZ7M6WJM2O3TI4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FRQASMNILNRZUZ7M6WJM2O3TI4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FRQASMNILNRZUZ7M6WJM2O3TI4/action/storage_attestation","attest_author":"https://pith.science/pith/FRQASMNILNRZUZ7M6WJM2O3TI4/action/author_attestation","sign_citation":"https://pith.science/pith/FRQASMNILNRZUZ7M6WJM2O3TI4/action/citation_signature","submit_replication":"https://pith.science/pith/FRQASMNILNRZUZ7M6WJM2O3TI4/action/replication_record"}},"created_at":"2026-07-05T11:28:13.378864+00:00","updated_at":"2026-07-05T11:28:13.378864+00:00"}