{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FGOTUAZU6MTUDF54UNDWTEBBMM","short_pith_number":"pith:FGOTUAZU","schema_version":"1.0","canonical_sha256":"299d3a0334f3274197bca3476990216314d5ad929a19479b3fd62f9a1fde2ead","source":{"kind":"arxiv","id":"2504.07440","version":3},"attestation_state":"computed","paper":{"title":"Model Utility Law: Evaluating LLMs beyond Performance through Mechanism Interpretable Metric","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiahao Ying, Xipeng Qiu, Xuanjing Huang, Yaoning Wang, Yixin Cao, Yugang Jiang","submitted_at":"2025-04-10T04:09:47Z","abstract_excerpt":"Large Language Models (LLMs) have become indispensable across academia, industry, and daily applications, yet current evaluation methods struggle to keep pace with their rapid development. One core challenge of evaluation in the large language model (LLM) era is the generalization issue: how to infer a model's near-unbounded abilities from inevitably bounded benchmarks. We address this challenge by proposing Model Utilization Index (MUI), a mechanism interpretability enhanced metric that complements traditional performance scores. MUI quantifies the effort a model expends on a task, defined as"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.07440","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-10T04:09:47Z","cross_cats_sorted":[],"title_canon_sha256":"76cf7a1b84732dc7717555998f6f5a7617183ba7cf8ac69269babee771f722e2","abstract_canon_sha256":"a6a0c6b1d5ba282940e456e6db2d104f632306ea3e566e1c31bd279f83374ecd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:46.491675Z","signature_b64":"mF2A2TdyZiuJe098+j+NTQB3c44SZviZ1riMkR0/F5pyjluYT56w+SXvunTiCkNeorxJuwmKA3aTaIIz558ZCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"299d3a0334f3274197bca3476990216314d5ad929a19479b3fd62f9a1fde2ead","last_reissued_at":"2026-07-05T11:09:46.491153Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:46.491153Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model Utility Law: Evaluating LLMs beyond Performance through Mechanism Interpretable Metric","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiahao Ying, Xipeng Qiu, Xuanjing Huang, Yaoning Wang, Yixin Cao, Yugang Jiang","submitted_at":"2025-04-10T04:09:47Z","abstract_excerpt":"Large Language Models (LLMs) have become indispensable across academia, industry, and daily applications, yet current evaluation methods struggle to keep pace with their rapid development. One core challenge of evaluation in the large language model (LLM) era is the generalization issue: how to infer a model's near-unbounded abilities from inevitably bounded benchmarks. We address this challenge by proposing Model Utilization Index (MUI), a mechanism interpretability enhanced metric that complements traditional performance scores. MUI quantifies the effort a model expends on a task, defined as"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.07440","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.07440/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.07440","created_at":"2026-07-05T11:09:46.491211+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.07440v3","created_at":"2026-07-05T11:09:46.491211+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.07440","created_at":"2026-07-05T11:09:46.491211+00:00"},{"alias_kind":"pith_short_12","alias_value":"FGOTUAZU6MTU","created_at":"2026-07-05T11:09:46.491211+00:00"},{"alias_kind":"pith_short_16","alias_value":"FGOTUAZU6MTUDF54","created_at":"2026-07-05T11:09:46.491211+00:00"},{"alias_kind":"pith_short_8","alias_value":"FGOTUAZU","created_at":"2026-07-05T11:09:46.491211+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22798","citing_title":"Does the Same Token Mean the Same State? MoE Routing as Signal for Reasoning Control","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02460","citing_title":"Neuron-Aware Data Selection for Annotation-Free LLM Self-Distillation","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FGOTUAZU6MTUDF54UNDWTEBBMM","json":"https://pith.science/pith/FGOTUAZU6MTUDF54UNDWTEBBMM.json","graph_json":"https://pith.science/api/pith-number/FGOTUAZU6MTUDF54UNDWTEBBMM/graph.json","events_json":"https://pith.science/api/pith-number/FGOTUAZU6MTUDF54UNDWTEBBMM/events.json","paper":"https://pith.science/paper/FGOTUAZU"},"agent_actions":{"view_html":"https://pith.science/pith/FGOTUAZU6MTUDF54UNDWTEBBMM","download_json":"https://pith.science/pith/FGOTUAZU6MTUDF54UNDWTEBBMM.json","view_paper":"https://pith.science/paper/FGOTUAZU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.07440&json=true","fetch_graph":"https://pith.science/api/pith-number/FGOTUAZU6MTUDF54UNDWTEBBMM/graph.json","fetch_events":"https://pith.science/api/pith-number/FGOTUAZU6MTUDF54UNDWTEBBMM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FGOTUAZU6MTUDF54UNDWTEBBMM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FGOTUAZU6MTUDF54UNDWTEBBMM/action/storage_attestation","attest_author":"https://pith.science/pith/FGOTUAZU6MTUDF54UNDWTEBBMM/action/author_attestation","sign_citation":"https://pith.science/pith/FGOTUAZU6MTUDF54UNDWTEBBMM/action/citation_signature","submit_replication":"https://pith.science/pith/FGOTUAZU6MTUDF54UNDWTEBBMM/action/replication_record"}},"created_at":"2026-07-05T11:09:46.491211+00:00","updated_at":"2026-07-05T11:09:46.491211+00:00"}