{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XM4DGYHBCK2JY5BZ6BCTL67NOB","short_pith_number":"pith:XM4DGYHB","schema_version":"1.0","canonical_sha256":"bb383360e112b49c7439f04535fbed704daf2efdb6a059c477e4ccf56c81aa2b","source":{"kind":"arxiv","id":"2404.11502","version":1},"attestation_state":"computed","paper":{"title":"Towards Coarse-to-Fine Evaluation of Inference Efficiency for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Erge Xiang, Jing Wang, Ji-Rong Wen, Linjiang Li, Tianyi Tang, Wayne Xin Zhao, Yunpeng Chai, Yushuo Chen","submitted_at":"2024-04-17T15:57:50Z","abstract_excerpt":"In real world, large language models (LLMs) can serve as the assistant to help users accomplish their jobs, and also support the development of advanced applications. For the wide application of LLMs, the inference efficiency is an essential concern, which has been widely studied in existing work, and numerous optimization algorithms and code libraries have been proposed to improve it. Nonetheless, users still find it challenging to compare the effectiveness of all the above methods and understand the underlying mechanisms. In this work, we perform a detailed coarse-to-fine analysis of the inf"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.11502","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-17T15:57:50Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3c122d135ea7a6bf0a64a9948d46de1f344789129dcbdafdcfa90f0f7a4d9b82","abstract_canon_sha256":"d62e352a1b966631a1c46a64f33779f434d47e477deb307f5d82108918b41af8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:09:09.656847Z","signature_b64":"ogMGDcC8wJHl7vDWaLVeo1+DLdelAr2pK2WZFWPPSIAFVMTw7bhQXC9TmN0u5GSd3bUVxFm7tBZ57wFoLiMfDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb383360e112b49c7439f04535fbed704daf2efdb6a059c477e4ccf56c81aa2b","last_reissued_at":"2026-07-05T08:09:09.656383Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:09:09.656383Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Coarse-to-Fine Evaluation of Inference Efficiency for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Erge Xiang, Jing Wang, Ji-Rong Wen, Linjiang Li, Tianyi Tang, Wayne Xin Zhao, Yunpeng Chai, Yushuo Chen","submitted_at":"2024-04-17T15:57:50Z","abstract_excerpt":"In real world, large language models (LLMs) can serve as the assistant to help users accomplish their jobs, and also support the development of advanced applications. For the wide application of LLMs, the inference efficiency is an essential concern, which has been widely studied in existing work, and numerous optimization algorithms and code libraries have been proposed to improve it. Nonetheless, users still find it challenging to compare the effectiveness of all the above methods and understand the underlying mechanisms. In this work, we perform a detailed coarse-to-fine analysis of the inf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.11502","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.11502/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.11502","created_at":"2026-07-05T08:09:09.656440+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.11502v1","created_at":"2026-07-05T08:09:09.656440+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.11502","created_at":"2026-07-05T08:09:09.656440+00:00"},{"alias_kind":"pith_short_12","alias_value":"XM4DGYHBCK2J","created_at":"2026-07-05T08:09:09.656440+00:00"},{"alias_kind":"pith_short_16","alias_value":"XM4DGYHBCK2JY5BZ","created_at":"2026-07-05T08:09:09.656440+00:00"},{"alias_kind":"pith_short_8","alias_value":"XM4DGYHB","created_at":"2026-07-05T08:09:09.656440+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10544","citing_title":"From Stacks to Circuits: A Regenerative Socio-Technical Roadmap for AI Infrastructure within Planetary Boundaries","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XM4DGYHBCK2JY5BZ6BCTL67NOB","json":"https://pith.science/pith/XM4DGYHBCK2JY5BZ6BCTL67NOB.json","graph_json":"https://pith.science/api/pith-number/XM4DGYHBCK2JY5BZ6BCTL67NOB/graph.json","events_json":"https://pith.science/api/pith-number/XM4DGYHBCK2JY5BZ6BCTL67NOB/events.json","paper":"https://pith.science/paper/XM4DGYHB"},"agent_actions":{"view_html":"https://pith.science/pith/XM4DGYHBCK2JY5BZ6BCTL67NOB","download_json":"https://pith.science/pith/XM4DGYHBCK2JY5BZ6BCTL67NOB.json","view_paper":"https://pith.science/paper/XM4DGYHB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.11502&json=true","fetch_graph":"https://pith.science/api/pith-number/XM4DGYHBCK2JY5BZ6BCTL67NOB/graph.json","fetch_events":"https://pith.science/api/pith-number/XM4DGYHBCK2JY5BZ6BCTL67NOB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XM4DGYHBCK2JY5BZ6BCTL67NOB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XM4DGYHBCK2JY5BZ6BCTL67NOB/action/storage_attestation","attest_author":"https://pith.science/pith/XM4DGYHBCK2JY5BZ6BCTL67NOB/action/author_attestation","sign_citation":"https://pith.science/pith/XM4DGYHBCK2JY5BZ6BCTL67NOB/action/citation_signature","submit_replication":"https://pith.science/pith/XM4DGYHBCK2JY5BZ6BCTL67NOB/action/replication_record"}},"created_at":"2026-07-05T08:09:09.656440+00:00","updated_at":"2026-07-05T08:09:09.656440+00:00"}