{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4XQNXXXKDWR2W3X2NURVZJXZLU","short_pith_number":"pith:4XQNXXXK","schema_version":"1.0","canonical_sha256":"e5e0dbdeea1da3ab6efa6d235ca6f95d0778e3402c37113e7507b74ebbc8569c","source":{"kind":"arxiv","id":"2507.09019","version":1},"attestation_state":"computed","paper":{"title":"On Evaluating Performance of LLM Inference Serving Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.DC"],"primary_cat":"cs.LG","authors_text":"Alexey Tumanov, Amey Agrawal, Anmol Agarwal, Jayashree Mohan, Nipun Kwatra, Nitin Kedia, Ramachandran Ramjee, Souvik Kundu","submitted_at":"2025-07-11T20:58:21Z","abstract_excerpt":"The rapid evolution of Large Language Model (LLM) inference systems has yielded significant efficiency improvements. However, our systematic analysis reveals that current evaluation methodologies frequently exhibit fundamental flaws, often manifesting as common evaluation anti-patterns that obscure true performance characteristics and impede scientific progress. Through a comprehensive examination of recent systems, we identify recurring anti-patterns across three key dimensions: Baseline Fairness, Evaluation Setup, and Metric Design. These anti-patterns are uniquely problematic for LLM infere"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.09019","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-11T20:58:21Z","cross_cats_sorted":["cs.AI","cs.DC"],"title_canon_sha256":"918d22784832be978cf053a6421515b3993c750067d1ecc29f2d3454cadc03b0","abstract_canon_sha256":"a30e47eb04f0b1e5808b7b3cdea352710526e0bf0d476af904ea788d2db5fa42"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:02.274923Z","signature_b64":"Pimfa0mHNwka4gD5b7oxKdlB8CDdFVFeOTM03ADI1uat5T6+9CZvX5uLI4XJxafWRHwCjac9BkbNj/rAhFCeAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e5e0dbdeea1da3ab6efa6d235ca6f95d0778e3402c37113e7507b74ebbc8569c","last_reissued_at":"2026-07-05T11:36:02.274479Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:02.274479Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Evaluating Performance of LLM Inference Serving Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.DC"],"primary_cat":"cs.LG","authors_text":"Alexey Tumanov, Amey Agrawal, Anmol Agarwal, Jayashree Mohan, Nipun Kwatra, Nitin Kedia, Ramachandran Ramjee, Souvik Kundu","submitted_at":"2025-07-11T20:58:21Z","abstract_excerpt":"The rapid evolution of Large Language Model (LLM) inference systems has yielded significant efficiency improvements. However, our systematic analysis reveals that current evaluation methodologies frequently exhibit fundamental flaws, often manifesting as common evaluation anti-patterns that obscure true performance characteristics and impede scientific progress. Through a comprehensive examination of recent systems, we identify recurring anti-patterns across three key dimensions: Baseline Fairness, Evaluation Setup, and Metric Design. These anti-patterns are uniquely problematic for LLM infere"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.09019","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.09019/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.09019","created_at":"2026-07-05T11:36:02.274536+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.09019v1","created_at":"2026-07-05T11:36:02.274536+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.09019","created_at":"2026-07-05T11:36:02.274536+00:00"},{"alias_kind":"pith_short_12","alias_value":"4XQNXXXKDWR2","created_at":"2026-07-05T11:36:02.274536+00:00"},{"alias_kind":"pith_short_16","alias_value":"4XQNXXXKDWR2W3X2","created_at":"2026-07-05T11:36:02.274536+00:00"},{"alias_kind":"pith_short_8","alias_value":"4XQNXXXK","created_at":"2026-07-05T11:36:02.274536+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29639","citing_title":"RTP-LLM: High-Performance Alibaba LLM Inference Engine","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01785","citing_title":"Seeing is Coding: On the Effectiveness of Vision Language Models in Code Understanding","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09611","citing_title":"Characterizing Performance-Energy Trade-offs of Large Language Models in Multi-Request Workflows","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15583","citing_title":"SAGE: Selective Attention-Guided Extraction for Token-Efficient Document Indexing","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4XQNXXXKDWR2W3X2NURVZJXZLU","json":"https://pith.science/pith/4XQNXXXKDWR2W3X2NURVZJXZLU.json","graph_json":"https://pith.science/api/pith-number/4XQNXXXKDWR2W3X2NURVZJXZLU/graph.json","events_json":"https://pith.science/api/pith-number/4XQNXXXKDWR2W3X2NURVZJXZLU/events.json","paper":"https://pith.science/paper/4XQNXXXK"},"agent_actions":{"view_html":"https://pith.science/pith/4XQNXXXKDWR2W3X2NURVZJXZLU","download_json":"https://pith.science/pith/4XQNXXXKDWR2W3X2NURVZJXZLU.json","view_paper":"https://pith.science/paper/4XQNXXXK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.09019&json=true","fetch_graph":"https://pith.science/api/pith-number/4XQNXXXKDWR2W3X2NURVZJXZLU/graph.json","fetch_events":"https://pith.science/api/pith-number/4XQNXXXKDWR2W3X2NURVZJXZLU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4XQNXXXKDWR2W3X2NURVZJXZLU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4XQNXXXKDWR2W3X2NURVZJXZLU/action/storage_attestation","attest_author":"https://pith.science/pith/4XQNXXXKDWR2W3X2NURVZJXZLU/action/author_attestation","sign_citation":"https://pith.science/pith/4XQNXXXKDWR2W3X2NURVZJXZLU/action/citation_signature","submit_replication":"https://pith.science/pith/4XQNXXXKDWR2W3X2NURVZJXZLU/action/replication_record"}},"created_at":"2026-07-05T11:36:02.274536+00:00","updated_at":"2026-07-05T11:36:02.274536+00:00"}