{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:SB6SWXTSWPUI75XP5NSDGPNQMY","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"bec15d190a8e67a33b721d596bbb9270011ef1cf2010cfc2b74bbcecd32c2eba","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-06-07T10:36:20Z","title_canon_sha256":"d3a8526042700cec6e919bbdae79f2feda50f4cea0d228feefb19b8e32190856"},"schema_version":"1.0","source":{"id":"2606.08562","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2606.08562","created_at":"2026-06-09T01:05:40Z"},{"alias_kind":"arxiv_version","alias_value":"2606.08562v1","created_at":"2026-06-09T01:05:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.08562","created_at":"2026-06-09T01:05:40Z"},{"alias_kind":"pith_short_12","alias_value":"SB6SWXTSWPUI","created_at":"2026-06-09T01:05:40Z"},{"alias_kind":"pith_short_16","alias_value":"SB6SWXTSWPUI75XP","created_at":"2026-06-09T01:05:40Z"},{"alias_kind":"pith_short_8","alias_value":"SB6SWXTS","created_at":"2026-06-09T01:05:40Z"}],"graph_snapshots":[{"event_id":"sha256:7752834ab54fe085eee9d7eba01d20791871abe5adcb29c3860441a0178f14dd","target":"graph","created_at":"2026-06-09T01:05:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2606.08562/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Transformer language models process input provided as subword fragments, but natural language semantics usually rely on word-level concepts. Detokenization is the process where models reconcile these two facts, aggregating subwords into word-level representations through their computation. Prior work has found that this takes place mostly in early-to-middle layers, but so far the exact mechanics of the process have not been pinned down. We venture deep into detokenization using activation patching in controlled paired experiments that isolate the contribution of different model components, loc","authors_text":"Benzi Busigin, Yuval Pinter","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-06-07T10:36:20Z","title":"Inside the LLM Word Factory"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.08562","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4578136fdd803fdbf338708138632e60fa9e2781112fea6cc302f9dfe7198590","target":"record","created_at":"2026-06-09T01:05:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"bec15d190a8e67a33b721d596bbb9270011ef1cf2010cfc2b74bbcecd32c2eba","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-06-07T10:36:20Z","title_canon_sha256":"d3a8526042700cec6e919bbdae79f2feda50f4cea0d228feefb19b8e32190856"},"schema_version":"1.0","source":{"id":"2606.08562","kind":"arxiv","version":1}},"canonical_sha256":"907d2b5e72b3e88ff6efeb64333db066358f571575ccac76d09f98336e46f9f6","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"907d2b5e72b3e88ff6efeb64333db066358f571575ccac76d09f98336e46f9f6","first_computed_at":"2026-06-09T01:05:40.200146Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-09T01:05:40.200146Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"s0nlyd4D+4X+veAGhYXv+OvcSNSgRCW0N7qDS2wVp5G8K/dR49TeQInjWMxWDdyNoxBxBxByyaaWOg8/RpNGDw==","signature_status":"signed_v1","signed_at":"2026-06-09T01:05:40.200572Z","signed_message":"canonical_sha256_bytes"},"source_id":"2606.08562","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4578136fdd803fdbf338708138632e60fa9e2781112fea6cc302f9dfe7198590","sha256:7752834ab54fe085eee9d7eba01d20791871abe5adcb29c3860441a0178f14dd"],"state_sha256":"a1fc8f0ad2a47ce981995c674b044b727638eee92cb4b17ff01b42b2a842dcf6"}