{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TEQEXZDL5AD5VQJPNZYESBB3A2","short_pith_number":"pith:TEQEXZDL","schema_version":"1.0","canonical_sha256":"99204be46be807dac12f6e7049043b069cc5d40f81a7b315222308a737299d5f","source":{"kind":"arxiv","id":"2503.01714","version":1},"attestation_state":"computed","paper":{"title":"Word Form Matters: LLMs' Semantic Reconstruction under Typoglycemia","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chenxi Wang, Lang Gao, Tianle Gu, Xiuying Chen, Zhongyu Wei, Zirui Song","submitted_at":"2025-03-03T16:31:45Z","abstract_excerpt":"Human readers can efficiently comprehend scrambled words, a phenomenon known as Typoglycemia, primarily by relying on word form; if word form alone is insufficient, they further utilize contextual cues for interpretation. While advanced large language models (LLMs) exhibit similar abilities, the underlying mechanisms remain unclear. To investigate this, we conduct controlled experiments to analyze the roles of word form and contextual information in semantic reconstruction and examine LLM attention patterns. Specifically, we first propose SemRecScore, a reliable metric to quantify the degree o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.01714","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-03T16:31:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d9bcce537d50b26a4f4f4641d9fc511df94ef40ee394e09383851c070d351bdd","abstract_canon_sha256":"96b59f76f0500366a96cb8cede692eb9a2587b1e6e8369c28032970926afe1fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:23:13.425585Z","signature_b64":"8H71l/Mllc0YspBmH0GyFruC17KXxSgCdY2BkLhf7NSlgoOPdxy33crviyNTSf0fpObzuYCOto+Ol/cBwTZfBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"99204be46be807dac12f6e7049043b069cc5d40f81a7b315222308a737299d5f","last_reissued_at":"2026-07-05T10:23:13.424938Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:23:13.424938Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Word Form Matters: LLMs' Semantic Reconstruction under Typoglycemia","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chenxi Wang, Lang Gao, Tianle Gu, Xiuying Chen, Zhongyu Wei, Zirui Song","submitted_at":"2025-03-03T16:31:45Z","abstract_excerpt":"Human readers can efficiently comprehend scrambled words, a phenomenon known as Typoglycemia, primarily by relying on word form; if word form alone is insufficient, they further utilize contextual cues for interpretation. While advanced large language models (LLMs) exhibit similar abilities, the underlying mechanisms remain unclear. To investigate this, we conduct controlled experiments to analyze the roles of word form and contextual information in semantic reconstruction and examine LLM attention patterns. Specifically, we first propose SemRecScore, a reliable metric to quantify the degree o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.01714","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.01714/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.01714","created_at":"2026-07-05T10:23:13.425027+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.01714v1","created_at":"2026-07-05T10:23:13.425027+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.01714","created_at":"2026-07-05T10:23:13.425027+00:00"},{"alias_kind":"pith_short_12","alias_value":"TEQEXZDL5AD5","created_at":"2026-07-05T10:23:13.425027+00:00"},{"alias_kind":"pith_short_16","alias_value":"TEQEXZDL5AD5VQJP","created_at":"2026-07-05T10:23:13.425027+00:00"},{"alias_kind":"pith_short_8","alias_value":"TEQEXZDL","created_at":"2026-07-05T10:23:13.425027+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.10641","citing_title":"Spelling-out is not Straightforward: LLMs' Capability of Tokenization from Token to Characters","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TEQEXZDL5AD5VQJPNZYESBB3A2","json":"https://pith.science/pith/TEQEXZDL5AD5VQJPNZYESBB3A2.json","graph_json":"https://pith.science/api/pith-number/TEQEXZDL5AD5VQJPNZYESBB3A2/graph.json","events_json":"https://pith.science/api/pith-number/TEQEXZDL5AD5VQJPNZYESBB3A2/events.json","paper":"https://pith.science/paper/TEQEXZDL"},"agent_actions":{"view_html":"https://pith.science/pith/TEQEXZDL5AD5VQJPNZYESBB3A2","download_json":"https://pith.science/pith/TEQEXZDL5AD5VQJPNZYESBB3A2.json","view_paper":"https://pith.science/paper/TEQEXZDL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.01714&json=true","fetch_graph":"https://pith.science/api/pith-number/TEQEXZDL5AD5VQJPNZYESBB3A2/graph.json","fetch_events":"https://pith.science/api/pith-number/TEQEXZDL5AD5VQJPNZYESBB3A2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TEQEXZDL5AD5VQJPNZYESBB3A2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TEQEXZDL5AD5VQJPNZYESBB3A2/action/storage_attestation","attest_author":"https://pith.science/pith/TEQEXZDL5AD5VQJPNZYESBB3A2/action/author_attestation","sign_citation":"https://pith.science/pith/TEQEXZDL5AD5VQJPNZYESBB3A2/action/citation_signature","submit_replication":"https://pith.science/pith/TEQEXZDL5AD5VQJPNZYESBB3A2/action/replication_record"}},"created_at":"2026-07-05T10:23:13.425027+00:00","updated_at":"2026-07-05T10:23:13.425027+00:00"}