{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RPEJLCIYTBWZ5K2RSOSX23NP4O","short_pith_number":"pith:RPEJLCIY","schema_version":"1.0","canonical_sha256":"8bc8958918986d9eab5193a57d6dafe3b3da0b752bde3f216abd2d7190029c7c","source":{"kind":"arxiv","id":"2504.03338","version":3},"attestation_state":"computed","paper":{"title":"BabyLM's First Words: Word Segmentation as a Phonological Probing Task","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Paula Buttery, Z\\'ebulon Goriely","submitted_at":"2025-04-04T10:42:56Z","abstract_excerpt":"Language models provide a key framework for studying linguistic theories based on prediction, but phonological analysis using large language models (LLMs) is difficult; there are few phonological benchmarks beyond English and the standard input representation used in LLMs (subwords of graphemes) is not suitable for analyzing the representation of phonemes. In this work, we demonstrate how word segmentation can be used as a phonological probing task, allowing us to study the representations learned by phoneme-based language models trained on child-directed speech across 31 languages. Following "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.03338","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-04T10:42:56Z","cross_cats_sorted":[],"title_canon_sha256":"c55261705a3eca28e909d7b40d8f77b701259860255764f0e3f1c21a7ba68feb","abstract_canon_sha256":"4d2bd0e457756df8f1d898640b7962f347fda25b24b78503f20c8e2afdad2f50"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:11.185082Z","signature_b64":"ihDxpX8EqeUdESjzTUFIb+173rPiloV4m9S6HUvMTOU29A7acBeu05zM+B/TJXu9rnA716P1HhBUOsQqF+TLDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8bc8958918986d9eab5193a57d6dafe3b3da0b752bde3f216abd2d7190029c7c","last_reissued_at":"2026-07-05T11:20:11.184659Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:11.184659Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BabyLM's First Words: Word Segmentation as a Phonological Probing Task","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Paula Buttery, Z\\'ebulon Goriely","submitted_at":"2025-04-04T10:42:56Z","abstract_excerpt":"Language models provide a key framework for studying linguistic theories based on prediction, but phonological analysis using large language models (LLMs) is difficult; there are few phonological benchmarks beyond English and the standard input representation used in LLMs (subwords of graphemes) is not suitable for analyzing the representation of phonemes. In this work, we demonstrate how word segmentation can be used as a phonological probing task, allowing us to study the representations learned by phoneme-based language models trained on child-directed speech across 31 languages. Following "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.03338","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.03338/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.03338","created_at":"2026-07-05T11:20:11.184716+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.03338v3","created_at":"2026-07-05T11:20:11.184716+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.03338","created_at":"2026-07-05T11:20:11.184716+00:00"},{"alias_kind":"pith_short_12","alias_value":"RPEJLCIYTBWZ","created_at":"2026-07-05T11:20:11.184716+00:00"},{"alias_kind":"pith_short_16","alias_value":"RPEJLCIYTBWZ5K2R","created_at":"2026-07-05T11:20:11.184716+00:00"},{"alias_kind":"pith_short_8","alias_value":"RPEJLCIY","created_at":"2026-07-05T11:20:11.184716+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.18639","citing_title":"ByteSpan: Information-Driven Subword Tokenisation","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RPEJLCIYTBWZ5K2RSOSX23NP4O","json":"https://pith.science/pith/RPEJLCIYTBWZ5K2RSOSX23NP4O.json","graph_json":"https://pith.science/api/pith-number/RPEJLCIYTBWZ5K2RSOSX23NP4O/graph.json","events_json":"https://pith.science/api/pith-number/RPEJLCIYTBWZ5K2RSOSX23NP4O/events.json","paper":"https://pith.science/paper/RPEJLCIY"},"agent_actions":{"view_html":"https://pith.science/pith/RPEJLCIYTBWZ5K2RSOSX23NP4O","download_json":"https://pith.science/pith/RPEJLCIYTBWZ5K2RSOSX23NP4O.json","view_paper":"https://pith.science/paper/RPEJLCIY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.03338&json=true","fetch_graph":"https://pith.science/api/pith-number/RPEJLCIYTBWZ5K2RSOSX23NP4O/graph.json","fetch_events":"https://pith.science/api/pith-number/RPEJLCIYTBWZ5K2RSOSX23NP4O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RPEJLCIYTBWZ5K2RSOSX23NP4O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RPEJLCIYTBWZ5K2RSOSX23NP4O/action/storage_attestation","attest_author":"https://pith.science/pith/RPEJLCIYTBWZ5K2RSOSX23NP4O/action/author_attestation","sign_citation":"https://pith.science/pith/RPEJLCIYTBWZ5K2RSOSX23NP4O/action/citation_signature","submit_replication":"https://pith.science/pith/RPEJLCIYTBWZ5K2RSOSX23NP4O/action/replication_record"}},"created_at":"2026-07-05T11:20:11.184716+00:00","updated_at":"2026-07-05T11:20:11.184716+00:00"}