{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NMECJJPHBGWZIH4GXXS24B2GPL","short_pith_number":"pith:NMECJJPH","schema_version":"1.0","canonical_sha256":"6b0824a5e709ad941f86bde5ae07467acf4c52bb453dfb8a48cf2504be74057e","source":{"kind":"arxiv","id":"2504.12459","version":1},"attestation_state":"computed","paper":{"title":"On Linear Representations and Pretraining Data Frequency in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jack Merullo, Noah A. Smith, Sarah Wiegreffe, Yanai Elazar","submitted_at":"2025-04-16T19:50:03Z","abstract_excerpt":"Pretraining data has a direct impact on the behaviors and quality of language models (LMs), but we only understand the most basic principles of this relationship. While most work focuses on pretraining data's effect on downstream task behavior, we investigate its relationship to LM representations. Previous work has discovered that, in language models, some concepts are encoded `linearly' in the representations, but what factors cause these representations to form? We study the connection between pretraining data frequency and models' linear representations of factual relations. We find eviden"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.12459","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-16T19:50:03Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5f9420c11601c55cca22dc04d691f283567e04419c06fe1e9969749abc3aca20","abstract_canon_sha256":"af39f113b1c6b6133df4b7307c6025b4bcc3cff55a0cfe7a9607ec88fd7efd17"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:16.728230Z","signature_b64":"55DgSI+h3WXEKatNgvAEeLwpzbQxd/iJZQzOImJ6N886HSYz55PpNlhHZN25cO0T2jeXkO9GPbBX/pVYCr1SAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6b0824a5e709ad941f86bde5ae07467acf4c52bb453dfb8a48cf2504be74057e","last_reissued_at":"2026-07-05T10:50:16.727693Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:16.727693Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Linear Representations and Pretraining Data Frequency in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jack Merullo, Noah A. Smith, Sarah Wiegreffe, Yanai Elazar","submitted_at":"2025-04-16T19:50:03Z","abstract_excerpt":"Pretraining data has a direct impact on the behaviors and quality of language models (LMs), but we only understand the most basic principles of this relationship. While most work focuses on pretraining data's effect on downstream task behavior, we investigate its relationship to LM representations. Previous work has discovered that, in language models, some concepts are encoded `linearly' in the representations, but what factors cause these representations to form? We study the connection between pretraining data frequency and models' linear representations of factual relations. We find eviden"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.12459","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.12459/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.12459","created_at":"2026-07-05T10:50:16.727756+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.12459v1","created_at":"2026-07-05T10:50:16.727756+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.12459","created_at":"2026-07-05T10:50:16.727756+00:00"},{"alias_kind":"pith_short_12","alias_value":"NMECJJPHBGWZ","created_at":"2026-07-05T10:50:16.727756+00:00"},{"alias_kind":"pith_short_16","alias_value":"NMECJJPHBGWZIH4G","created_at":"2026-07-05T10:50:16.727756+00:00"},{"alias_kind":"pith_short_8","alias_value":"NMECJJPH","created_at":"2026-07-05T10:50:16.727756+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.01685","citing_title":"How Do Language Models Compose Functions?","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12412","citing_title":"Stories in Space: In-Context Learning Trajectories in Conceptual Belief Space","ref_index":150,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NMECJJPHBGWZIH4GXXS24B2GPL","json":"https://pith.science/pith/NMECJJPHBGWZIH4GXXS24B2GPL.json","graph_json":"https://pith.science/api/pith-number/NMECJJPHBGWZIH4GXXS24B2GPL/graph.json","events_json":"https://pith.science/api/pith-number/NMECJJPHBGWZIH4GXXS24B2GPL/events.json","paper":"https://pith.science/paper/NMECJJPH"},"agent_actions":{"view_html":"https://pith.science/pith/NMECJJPHBGWZIH4GXXS24B2GPL","download_json":"https://pith.science/pith/NMECJJPHBGWZIH4GXXS24B2GPL.json","view_paper":"https://pith.science/paper/NMECJJPH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.12459&json=true","fetch_graph":"https://pith.science/api/pith-number/NMECJJPHBGWZIH4GXXS24B2GPL/graph.json","fetch_events":"https://pith.science/api/pith-number/NMECJJPHBGWZIH4GXXS24B2GPL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NMECJJPHBGWZIH4GXXS24B2GPL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NMECJJPHBGWZIH4GXXS24B2GPL/action/storage_attestation","attest_author":"https://pith.science/pith/NMECJJPHBGWZIH4GXXS24B2GPL/action/author_attestation","sign_citation":"https://pith.science/pith/NMECJJPHBGWZIH4GXXS24B2GPL/action/citation_signature","submit_replication":"https://pith.science/pith/NMECJJPHBGWZIH4GXXS24B2GPL/action/replication_record"}},"created_at":"2026-07-05T10:50:16.727756+00:00","updated_at":"2026-07-05T10:50:16.727756+00:00"}