{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FJNOBFZP2KPATXDKZEQVSDDAOU","short_pith_number":"pith:FJNOBFZP","schema_version":"1.0","canonical_sha256":"2a5ae0972fd29e09dc6ac921590c6075295c3bfd73d0a00a395b240a5818a691","source":{"kind":"arxiv","id":"2410.04335","version":1},"attestation_state":"computed","paper":{"title":"ReTok: Replacing Tokenizer to Enhance Representation Efficiency in Large Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bowen Zhang, Guang Liu, Jijie Li, Liangdong Wang, Mengdi Zhao, Shuhao Gu","submitted_at":"2024-10-06T03:01:07Z","abstract_excerpt":"Tokenizer is an essential component for large language models (LLMs), and a tokenizer with a high compression rate can improve the model's representation and processing efficiency. However, the tokenizer cannot ensure high compression rate in all scenarios, and an increase in the average input and output lengths will increases the training and inference costs of the model. Therefore, it is crucial to find ways to improve the model's efficiency with minimal cost while maintaining the model's performance. In this work, we propose a method to improve model representation and processing efficiency"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.04335","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-06T03:01:07Z","cross_cats_sorted":[],"title_canon_sha256":"b30657d888adc10cc9fd8bc5a5d8e94115842f5543da8fd6a9ee0a1c5937d769","abstract_canon_sha256":"e28b832d14321746bdcfe9dc5cbf959f5e3718e6d8ffb08bff453321594abe1c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:32.928674Z","signature_b64":"SX88hAleeiOXFIafBYoVszoZTeqq16wopYwe++sQuxj8lkP7pJy864paqmbARc1cML8eaxxOwS3d5DIdQzqfAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a5ae0972fd29e09dc6ac921590c6075295c3bfd73d0a00a395b240a5818a691","last_reissued_at":"2026-07-05T09:16:32.928214Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:32.928214Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReTok: Replacing Tokenizer to Enhance Representation Efficiency in Large Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bowen Zhang, Guang Liu, Jijie Li, Liangdong Wang, Mengdi Zhao, Shuhao Gu","submitted_at":"2024-10-06T03:01:07Z","abstract_excerpt":"Tokenizer is an essential component for large language models (LLMs), and a tokenizer with a high compression rate can improve the model's representation and processing efficiency. However, the tokenizer cannot ensure high compression rate in all scenarios, and an increase in the average input and output lengths will increases the training and inference costs of the model. Therefore, it is crucial to find ways to improve the model's efficiency with minimal cost while maintaining the model's performance. In this work, we propose a method to improve model representation and processing efficiency"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.04335","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.04335/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.04335","created_at":"2026-07-05T09:16:32.928279+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.04335v1","created_at":"2026-07-05T09:16:32.928279+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.04335","created_at":"2026-07-05T09:16:32.928279+00:00"},{"alias_kind":"pith_short_12","alias_value":"FJNOBFZP2KPA","created_at":"2026-07-05T09:16:32.928279+00:00"},{"alias_kind":"pith_short_16","alias_value":"FJNOBFZP2KPATXDK","created_at":"2026-07-05T09:16:32.928279+00:00"},{"alias_kind":"pith_short_8","alias_value":"FJNOBFZP","created_at":"2026-07-05T09:16:32.928279+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FJNOBFZP2KPATXDKZEQVSDDAOU","json":"https://pith.science/pith/FJNOBFZP2KPATXDKZEQVSDDAOU.json","graph_json":"https://pith.science/api/pith-number/FJNOBFZP2KPATXDKZEQVSDDAOU/graph.json","events_json":"https://pith.science/api/pith-number/FJNOBFZP2KPATXDKZEQVSDDAOU/events.json","paper":"https://pith.science/paper/FJNOBFZP"},"agent_actions":{"view_html":"https://pith.science/pith/FJNOBFZP2KPATXDKZEQVSDDAOU","download_json":"https://pith.science/pith/FJNOBFZP2KPATXDKZEQVSDDAOU.json","view_paper":"https://pith.science/paper/FJNOBFZP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.04335&json=true","fetch_graph":"https://pith.science/api/pith-number/FJNOBFZP2KPATXDKZEQVSDDAOU/graph.json","fetch_events":"https://pith.science/api/pith-number/FJNOBFZP2KPATXDKZEQVSDDAOU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FJNOBFZP2KPATXDKZEQVSDDAOU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FJNOBFZP2KPATXDKZEQVSDDAOU/action/storage_attestation","attest_author":"https://pith.science/pith/FJNOBFZP2KPATXDKZEQVSDDAOU/action/author_attestation","sign_citation":"https://pith.science/pith/FJNOBFZP2KPATXDKZEQVSDDAOU/action/citation_signature","submit_replication":"https://pith.science/pith/FJNOBFZP2KPATXDKZEQVSDDAOU/action/replication_record"}},"created_at":"2026-07-05T09:16:32.928279+00:00","updated_at":"2026-07-05T09:16:32.928279+00:00"}