{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7QY5GLA7G5FFWDMXSMDBRV4J5A","short_pith_number":"pith:7QY5GLA7","schema_version":"1.0","canonical_sha256":"fc31d32c1f374a5b0d97930618d789e8374300a1fb1b873b3f1e22c216bdd1c6","source":{"kind":"arxiv","id":"2412.11494","version":1},"attestation_state":"computed","paper":{"title":"FTP: A Fine-grained Token-wise Pruner for Large Language Models via Token Routing","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dong Li, Emad Barsoum, Fuwei Yang, Haiduo Huang, Han Liu, Haowei Zhu, Ji Liu, Jintu Zheng, Jinzhang Peng, Lu Tian, Zekai Li, Zeping Li","submitted_at":"2024-12-16T07:09:46Z","abstract_excerpt":"Recently, large language models (LLMs) have demonstrated superior performance across various tasks by adhering to scaling laws, which significantly increase model size. However, the huge computation overhead during inference hinders the deployment in industrial applications. Many works leverage traditional compression approaches to boost model inference, but these always introduce additional training costs to restore the performance and the pruning results typically show noticeable performance drops compared to the original model when aiming for a specific level of acceleration. To address the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.11494","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-16T07:09:46Z","cross_cats_sorted":[],"title_canon_sha256":"76f358b75c5690e25425eabd82189c9f8f055827cedbc5ac9abbaf348d8bd5d2","abstract_canon_sha256":"806c07c88c42dd54ee5ef5dcf93093a30de9f32820af7cbd3798c9afd91308c3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:49:33.948336Z","signature_b64":"jriQrZmRJ/3MB4X0jMcPiVvzX2bq8glLszPtUyUYrnZ7E4SOZse4SnyIgF390xnHnDo15YDfsRQ2M3EVOVNhBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fc31d32c1f374a5b0d97930618d789e8374300a1fb1b873b3f1e22c216bdd1c6","last_reissued_at":"2026-07-05T09:49:33.947876Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:49:33.947876Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FTP: A Fine-grained Token-wise Pruner for Large Language Models via Token Routing","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dong Li, Emad Barsoum, Fuwei Yang, Haiduo Huang, Han Liu, Haowei Zhu, Ji Liu, Jintu Zheng, Jinzhang Peng, Lu Tian, Zekai Li, Zeping Li","submitted_at":"2024-12-16T07:09:46Z","abstract_excerpt":"Recently, large language models (LLMs) have demonstrated superior performance across various tasks by adhering to scaling laws, which significantly increase model size. However, the huge computation overhead during inference hinders the deployment in industrial applications. Many works leverage traditional compression approaches to boost model inference, but these always introduce additional training costs to restore the performance and the pruning results typically show noticeable performance drops compared to the original model when aiming for a specific level of acceleration. To address the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.11494","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.11494/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.11494","created_at":"2026-07-05T09:49:33.947928+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.11494v1","created_at":"2026-07-05T09:49:33.947928+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.11494","created_at":"2026-07-05T09:49:33.947928+00:00"},{"alias_kind":"pith_short_12","alias_value":"7QY5GLA7G5FF","created_at":"2026-07-05T09:49:33.947928+00:00"},{"alias_kind":"pith_short_16","alias_value":"7QY5GLA7G5FFWDMX","created_at":"2026-07-05T09:49:33.947928+00:00"},{"alias_kind":"pith_short_8","alias_value":"7QY5GLA7","created_at":"2026-07-05T09:49:33.947928+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7QY5GLA7G5FFWDMXSMDBRV4J5A","json":"https://pith.science/pith/7QY5GLA7G5FFWDMXSMDBRV4J5A.json","graph_json":"https://pith.science/api/pith-number/7QY5GLA7G5FFWDMXSMDBRV4J5A/graph.json","events_json":"https://pith.science/api/pith-number/7QY5GLA7G5FFWDMXSMDBRV4J5A/events.json","paper":"https://pith.science/paper/7QY5GLA7"},"agent_actions":{"view_html":"https://pith.science/pith/7QY5GLA7G5FFWDMXSMDBRV4J5A","download_json":"https://pith.science/pith/7QY5GLA7G5FFWDMXSMDBRV4J5A.json","view_paper":"https://pith.science/paper/7QY5GLA7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.11494&json=true","fetch_graph":"https://pith.science/api/pith-number/7QY5GLA7G5FFWDMXSMDBRV4J5A/graph.json","fetch_events":"https://pith.science/api/pith-number/7QY5GLA7G5FFWDMXSMDBRV4J5A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7QY5GLA7G5FFWDMXSMDBRV4J5A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7QY5GLA7G5FFWDMXSMDBRV4J5A/action/storage_attestation","attest_author":"https://pith.science/pith/7QY5GLA7G5FFWDMXSMDBRV4J5A/action/author_attestation","sign_citation":"https://pith.science/pith/7QY5GLA7G5FFWDMXSMDBRV4J5A/action/citation_signature","submit_replication":"https://pith.science/pith/7QY5GLA7G5FFWDMXSMDBRV4J5A/action/replication_record"}},"created_at":"2026-07-05T09:49:33.947928+00:00","updated_at":"2026-07-05T09:49:33.947928+00:00"}