{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JPKZLI77CVLTF4AQQNBUYTDZOP","short_pith_number":"pith:JPKZLI77","schema_version":"1.0","canonical_sha256":"4bd595a3ff155732f01083434c4c7973fdc6d98642ae87837df242dfea0e5675","source":{"kind":"arxiv","id":"2507.14423","version":1},"attestation_state":"computed","paper":{"title":"On the Effect of Token Merging on Pre-trained Models for Code","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Ahmed E. Hassan, Hao Li, Mootez Saad, Tushar Sharma","submitted_at":"2025-07-19T00:48:20Z","abstract_excerpt":"Tokenization is a fundamental component of language models for code. It involves breaking down the input into units that are later passed to the language model stack to learn high-dimensional representations used in various contexts, from classification to generation. However, the output of these tokenizers is often longer than that traditionally used in compilers and interpreters. This could result in undesirable effects, such as increased computational overhead. In this work, we investigate the effect of merging the hidden representations of subtokens that belong to the same semantic unit, s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.14423","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-07-19T00:48:20Z","cross_cats_sorted":[],"title_canon_sha256":"bcfc93c17da8726f0238f2a65c8887376e26509a80eebdd7056030fd9fe10e95","abstract_canon_sha256":"a5b5e05ec559b65e0257bfe2dac7c5e1e7c03f1c9119821094833b153ca1752e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:06.278949Z","signature_b64":"CWFLySvzC6Zovz2HfIsx2tbJGRcsk8j2+AADiq9OocOWjP4y53Szqq6FMCJis2jj26w6GRXPWgAPM6dZSnuYBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4bd595a3ff155732f01083434c4c7973fdc6d98642ae87837df242dfea0e5675","last_reissued_at":"2026-07-05T11:40:06.278488Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:06.278488Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Effect of Token Merging on Pre-trained Models for Code","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Ahmed E. Hassan, Hao Li, Mootez Saad, Tushar Sharma","submitted_at":"2025-07-19T00:48:20Z","abstract_excerpt":"Tokenization is a fundamental component of language models for code. It involves breaking down the input into units that are later passed to the language model stack to learn high-dimensional representations used in various contexts, from classification to generation. However, the output of these tokenizers is often longer than that traditionally used in compilers and interpreters. This could result in undesirable effects, such as increased computational overhead. In this work, we investigate the effect of merging the hidden representations of subtokens that belong to the same semantic unit, s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.14423","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.14423/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.14423","created_at":"2026-07-05T11:40:06.278548+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.14423v1","created_at":"2026-07-05T11:40:06.278548+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.14423","created_at":"2026-07-05T11:40:06.278548+00:00"},{"alias_kind":"pith_short_12","alias_value":"JPKZLI77CVLT","created_at":"2026-07-05T11:40:06.278548+00:00"},{"alias_kind":"pith_short_16","alias_value":"JPKZLI77CVLTF4AQ","created_at":"2026-07-05T11:40:06.278548+00:00"},{"alias_kind":"pith_short_8","alias_value":"JPKZLI77","created_at":"2026-07-05T11:40:06.278548+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JPKZLI77CVLTF4AQQNBUYTDZOP","json":"https://pith.science/pith/JPKZLI77CVLTF4AQQNBUYTDZOP.json","graph_json":"https://pith.science/api/pith-number/JPKZLI77CVLTF4AQQNBUYTDZOP/graph.json","events_json":"https://pith.science/api/pith-number/JPKZLI77CVLTF4AQQNBUYTDZOP/events.json","paper":"https://pith.science/paper/JPKZLI77"},"agent_actions":{"view_html":"https://pith.science/pith/JPKZLI77CVLTF4AQQNBUYTDZOP","download_json":"https://pith.science/pith/JPKZLI77CVLTF4AQQNBUYTDZOP.json","view_paper":"https://pith.science/paper/JPKZLI77","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.14423&json=true","fetch_graph":"https://pith.science/api/pith-number/JPKZLI77CVLTF4AQQNBUYTDZOP/graph.json","fetch_events":"https://pith.science/api/pith-number/JPKZLI77CVLTF4AQQNBUYTDZOP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JPKZLI77CVLTF4AQQNBUYTDZOP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JPKZLI77CVLTF4AQQNBUYTDZOP/action/storage_attestation","attest_author":"https://pith.science/pith/JPKZLI77CVLTF4AQQNBUYTDZOP/action/author_attestation","sign_citation":"https://pith.science/pith/JPKZLI77CVLTF4AQQNBUYTDZOP/action/citation_signature","submit_replication":"https://pith.science/pith/JPKZLI77CVLTF4AQQNBUYTDZOP/action/replication_record"}},"created_at":"2026-07-05T11:40:06.278548+00:00","updated_at":"2026-07-05T11:40:06.278548+00:00"}