{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NYK7EM2I4CJQQEVGD5CYM7KGO3","short_pith_number":"pith:NYK7EM2I","schema_version":"1.0","canonical_sha256":"6e15f23348e0930812a61f45867d4676ec912b6c2eca548d1bb919b70ba743a9","source":{"kind":"arxiv","id":"2504.17562","version":2},"attestation_state":"computed","paper":{"title":"When Does Metadata Conditioning (NOT) Work for Language Model Pre-Training? A Study with Context-Free Grammars","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Daisuke Okanohara, Kazusato Oko, Kohei Hayashi, Naoki Nishikawa, Rei Higuchi, Ryotaro Kawata, Seiya Tokui, Shoichiro Yamaguchi, Sosuke Kobayashi, Taiji Suzuki","submitted_at":"2025-04-24T13:56:43Z","abstract_excerpt":"The ability to acquire latent semantics is one of the key properties that determines the performance of language models. One convenient approach to invoke this ability is to prepend metadata (e.g. URLs, domains, and styles) at the beginning of texts in the pre-training data, making it easier for the model to access latent semantics before observing the entire text. Previous studies have reported that this technique actually improves the performance of trained models in downstream tasks; however, this improvement has been observed only in specific downstream tasks, without consistent enhancemen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.17562","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-24T13:56:43Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"8dcad6f7a815fd5cb1ad2cce5ebe1edb42d945e55001a15c2b10353ac56f52ed","abstract_canon_sha256":"4a7a8f745ea2d0c31b65a67d76980e0fe1dab7528f7604ead58a805019a16c9a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:44:04.746670Z","signature_b64":"MXujn5HepMxuoau4zax4nAEzV4Ae5r7PNqOXu9f4gR3mrT5W0vOP/YaFDsrBJl7w9Ad0tfrOsGkelEBD0HNMCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e15f23348e0930812a61f45867d4676ec912b6c2eca548d1bb919b70ba743a9","last_reissued_at":"2026-07-05T11:44:04.746180Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:44:04.746180Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Does Metadata Conditioning (NOT) Work for Language Model Pre-Training? A Study with Context-Free Grammars","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Daisuke Okanohara, Kazusato Oko, Kohei Hayashi, Naoki Nishikawa, Rei Higuchi, Ryotaro Kawata, Seiya Tokui, Shoichiro Yamaguchi, Sosuke Kobayashi, Taiji Suzuki","submitted_at":"2025-04-24T13:56:43Z","abstract_excerpt":"The ability to acquire latent semantics is one of the key properties that determines the performance of language models. One convenient approach to invoke this ability is to prepend metadata (e.g. URLs, domains, and styles) at the beginning of texts in the pre-training data, making it easier for the model to access latent semantics before observing the entire text. Previous studies have reported that this technique actually improves the performance of trained models in downstream tasks; however, this improvement has been observed only in specific downstream tasks, without consistent enhancemen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.17562","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.17562/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.17562","created_at":"2026-07-05T11:44:04.746240+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.17562v2","created_at":"2026-07-05T11:44:04.746240+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.17562","created_at":"2026-07-05T11:44:04.746240+00:00"},{"alias_kind":"pith_short_12","alias_value":"NYK7EM2I4CJQ","created_at":"2026-07-05T11:44:04.746240+00:00"},{"alias_kind":"pith_short_16","alias_value":"NYK7EM2I4CJQQEVG","created_at":"2026-07-05T11:44:04.746240+00:00"},{"alias_kind":"pith_short_8","alias_value":"NYK7EM2I","created_at":"2026-07-05T11:44:04.746240+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.21613","citing_title":"Beyond URLs: Metadata Diversity and Position for Efficient LLM Pretraining","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NYK7EM2I4CJQQEVGD5CYM7KGO3","json":"https://pith.science/pith/NYK7EM2I4CJQQEVGD5CYM7KGO3.json","graph_json":"https://pith.science/api/pith-number/NYK7EM2I4CJQQEVGD5CYM7KGO3/graph.json","events_json":"https://pith.science/api/pith-number/NYK7EM2I4CJQQEVGD5CYM7KGO3/events.json","paper":"https://pith.science/paper/NYK7EM2I"},"agent_actions":{"view_html":"https://pith.science/pith/NYK7EM2I4CJQQEVGD5CYM7KGO3","download_json":"https://pith.science/pith/NYK7EM2I4CJQQEVGD5CYM7KGO3.json","view_paper":"https://pith.science/paper/NYK7EM2I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.17562&json=true","fetch_graph":"https://pith.science/api/pith-number/NYK7EM2I4CJQQEVGD5CYM7KGO3/graph.json","fetch_events":"https://pith.science/api/pith-number/NYK7EM2I4CJQQEVGD5CYM7KGO3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NYK7EM2I4CJQQEVGD5CYM7KGO3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NYK7EM2I4CJQQEVGD5CYM7KGO3/action/storage_attestation","attest_author":"https://pith.science/pith/NYK7EM2I4CJQQEVGD5CYM7KGO3/action/author_attestation","sign_citation":"https://pith.science/pith/NYK7EM2I4CJQQEVGD5CYM7KGO3/action/citation_signature","submit_replication":"https://pith.science/pith/NYK7EM2I4CJQQEVGD5CYM7KGO3/action/replication_record"}},"created_at":"2026-07-05T11:44:04.746240+00:00","updated_at":"2026-07-05T11:44:04.746240+00:00"}