{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:U24MI4Y4NV52Q3XG5OYGFMEDQB","short_pith_number":"pith:U24MI4Y4","schema_version":"1.0","canonical_sha256":"a6b8c4731c6d7ba86ee6ebb062b083805ce332cf689723a080101d0878696c3b","source":{"kind":"arxiv","id":"2502.03275","version":2},"attestation_state":"computed","paper":{"title":"Token Assorted: Mixing Latent and Text Tokens for Improved Language Model Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.LO"],"primary_cat":"cs.CL","authors_text":"DiJia Su, Hanlin Zhu, Jiantao Jiao, Qinqing Zheng, Yingchen Xu, Yuandong Tian","submitted_at":"2025-02-05T15:33:00Z","abstract_excerpt":"Large Language Models (LLMs) excel at reasoning and planning when trained on chainof-thought (CoT) data, where the step-by-step thought process is explicitly outlined by text tokens. However, this results in lengthy inputs where many words support textual coherence rather than core reasoning information, and processing these inputs consumes substantial computation resources. In this work, we propose a hybrid representation of the reasoning process, where we partially abstract away the initial reasoning steps using latent discrete tokens generated by VQ-VAE, significantly reducing the length of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.03275","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-05T15:33:00Z","cross_cats_sorted":["cs.AI","cs.LG","cs.LO"],"title_canon_sha256":"f8d80571b5166576050c99b18ec306a0c042e9801906d332a7dfe0f0627bca93","abstract_canon_sha256":"147b06a3e05b0b8b528c11adc82343f6f32195f1722b9fe267ad2934e9d0ff23"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:02:40.260193Z","signature_b64":"6HEXfcENlRkfKG0p+hOiaKzIo/jh4AGYe8mI+stsAsYi0HPqDgIqPfqi43XipL2ia8trfBXx4j2H0oUbwCPHAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a6b8c4731c6d7ba86ee6ebb062b083805ce332cf689723a080101d0878696c3b","last_reissued_at":"2026-07-05T12:02:40.259697Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:02:40.259697Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Token Assorted: Mixing Latent and Text Tokens for Improved Language Model Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.LO"],"primary_cat":"cs.CL","authors_text":"DiJia Su, Hanlin Zhu, Jiantao Jiao, Qinqing Zheng, Yingchen Xu, Yuandong Tian","submitted_at":"2025-02-05T15:33:00Z","abstract_excerpt":"Large Language Models (LLMs) excel at reasoning and planning when trained on chainof-thought (CoT) data, where the step-by-step thought process is explicitly outlined by text tokens. However, this results in lengthy inputs where many words support textual coherence rather than core reasoning information, and processing these inputs consumes substantial computation resources. In this work, we propose a hybrid representation of the reasoning process, where we partially abstract away the initial reasoning steps using latent discrete tokens generated by VQ-VAE, significantly reducing the length of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.03275","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.03275/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.03275","created_at":"2026-07-05T12:02:40.259749+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.03275v2","created_at":"2026-07-05T12:02:40.259749+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.03275","created_at":"2026-07-05T12:02:40.259749+00:00"},{"alias_kind":"pith_short_12","alias_value":"U24MI4Y4NV52","created_at":"2026-07-05T12:02:40.259749+00:00"},{"alias_kind":"pith_short_16","alias_value":"U24MI4Y4NV52Q3XG","created_at":"2026-07-05T12:02:40.259749+00:00"},{"alias_kind":"pith_short_8","alias_value":"U24MI4Y4","created_at":"2026-07-05T12:02:40.259749+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":183,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17924","citing_title":"PearlVLA: Progressive Embodied Action-Plan Refinement in Latent Space","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07108","citing_title":"DyCon: Dynamic Reasoning Control via Evolving Difficulty Modeling","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05859","citing_title":"TARPO: Token-Wise Latent-Explicit Reasoning via Action-Routing Policy Optimization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02871","citing_title":"Adaptive Latent Agentic Reasoning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02248","citing_title":"Geometric Latent Reasoning Induces Shorter Generations in LLMs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02842","citing_title":"Spectral-Progressive Thought Flow for Lightweight Multimodal Reasoning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01532","citing_title":"Rethinking the Role of Positional Encoding: Sliding-Window Transformers without PE Remain Turing Complete","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29928","citing_title":"Latent-CURE for Breast Cancer Diagnosis","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28600","citing_title":"Transformers Provably Learn to Internalize Chain-of-Thought","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14323","citing_title":"Dynamic Latent Routing","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16419","citing_title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","ref_index":163,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26355","citing_title":"Shorthand for Thought: Compressing LLM Reasoning via Entropy-Guided Supertokens","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18486","citing_title":"Xiaomi OneVL: One-Step Latent Reasoning and Planning with Vision-Language Explanation","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06165","citing_title":"Post Reasoning: Improving the Performance of Non-Thinking Models at No Cost","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08299","citing_title":"SeLaR: Selective Latent Reasoning in Large Language Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18486","citing_title":"Xiaomi OneVL: One-Step Latent Reasoning and Planning with Vision-Language Explanation","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17892","citing_title":"LEPO: Latent Reasoning Policy Optimization for Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":86,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U24MI4Y4NV52Q3XG5OYGFMEDQB","json":"https://pith.science/pith/U24MI4Y4NV52Q3XG5OYGFMEDQB.json","graph_json":"https://pith.science/api/pith-number/U24MI4Y4NV52Q3XG5OYGFMEDQB/graph.json","events_json":"https://pith.science/api/pith-number/U24MI4Y4NV52Q3XG5OYGFMEDQB/events.json","paper":"https://pith.science/paper/U24MI4Y4"},"agent_actions":{"view_html":"https://pith.science/pith/U24MI4Y4NV52Q3XG5OYGFMEDQB","download_json":"https://pith.science/pith/U24MI4Y4NV52Q3XG5OYGFMEDQB.json","view_paper":"https://pith.science/paper/U24MI4Y4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.03275&json=true","fetch_graph":"https://pith.science/api/pith-number/U24MI4Y4NV52Q3XG5OYGFMEDQB/graph.json","fetch_events":"https://pith.science/api/pith-number/U24MI4Y4NV52Q3XG5OYGFMEDQB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U24MI4Y4NV52Q3XG5OYGFMEDQB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U24MI4Y4NV52Q3XG5OYGFMEDQB/action/storage_attestation","attest_author":"https://pith.science/pith/U24MI4Y4NV52Q3XG5OYGFMEDQB/action/author_attestation","sign_citation":"https://pith.science/pith/U24MI4Y4NV52Q3XG5OYGFMEDQB/action/citation_signature","submit_replication":"https://pith.science/pith/U24MI4Y4NV52Q3XG5OYGFMEDQB/action/replication_record"}},"created_at":"2026-07-05T12:02:40.259749+00:00","updated_at":"2026-07-05T12:02:40.259749+00:00"}