{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FQ32LH2OFET6F7WFBT36C7DVV6","short_pith_number":"pith:FQ32LH2O","schema_version":"1.0","canonical_sha256":"2c37a59f4e2927e2fec50cf7e17c75af9cc24cdbeef7ef8d895a7db9ef200ee4","source":{"kind":"arxiv","id":"2404.09135","version":1},"attestation_state":"computed","paper":{"title":"Unveiling LLM Evaluation Focused on Metrics: Challenges and Solutions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Taojun Hu, Xiao-Hua Zhou","submitted_at":"2024-04-14T03:54:00Z","abstract_excerpt":"Natural Language Processing (NLP) is witnessing a remarkable breakthrough driven by the success of Large Language Models (LLMs). LLMs have gained significant attention across academia and industry for their versatile applications in text generation, question answering, and text summarization. As the landscape of NLP evolves with an increasing number of domain-specific LLMs employing diverse techniques and trained on various corpus, evaluating performance of these models becomes paramount. To quantify the performance, it's crucial to have a comprehensive grasp of existing metrics. Among the eva"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.09135","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-14T03:54:00Z","cross_cats_sorted":[],"title_canon_sha256":"1d3c0b8754a216b7fa16abd04d718b3f896dda4aca129002b55cfd8edc10c542","abstract_canon_sha256":"49902c5bc07697e6f7b6e58f0b4f5f0e9e41e7da803fa2581477149b3253897e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:07:45.576500Z","signature_b64":"VszLnGryTQRhxjR3aXDlmjhlxaksp/uhjlzD6Xk0lzypbBwTtZPLNmLlFcKEZeUftJOUA6d1YCARrvLMD8zsCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c37a59f4e2927e2fec50cf7e17c75af9cc24cdbeef7ef8d895a7db9ef200ee4","last_reissued_at":"2026-07-05T08:07:45.576027Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:07:45.576027Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unveiling LLM Evaluation Focused on Metrics: Challenges and Solutions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Taojun Hu, Xiao-Hua Zhou","submitted_at":"2024-04-14T03:54:00Z","abstract_excerpt":"Natural Language Processing (NLP) is witnessing a remarkable breakthrough driven by the success of Large Language Models (LLMs). LLMs have gained significant attention across academia and industry for their versatile applications in text generation, question answering, and text summarization. As the landscape of NLP evolves with an increasing number of domain-specific LLMs employing diverse techniques and trained on various corpus, evaluating performance of these models becomes paramount. To quantify the performance, it's crucial to have a comprehensive grasp of existing metrics. Among the eva"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.09135","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.09135/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.09135","created_at":"2026-07-05T08:07:45.576082+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.09135v1","created_at":"2026-07-05T08:07:45.576082+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.09135","created_at":"2026-07-05T08:07:45.576082+00:00"},{"alias_kind":"pith_short_12","alias_value":"FQ32LH2OFET6","created_at":"2026-07-05T08:07:45.576082+00:00"},{"alias_kind":"pith_short_16","alias_value":"FQ32LH2OFET6F7WF","created_at":"2026-07-05T08:07:45.576082+00:00"},{"alias_kind":"pith_short_8","alias_value":"FQ32LH2O","created_at":"2026-07-05T08:07:45.576082+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01386","citing_title":"GuidaPA: Privacy-Preserving Chatbot for Public Administration via Federated Learning","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23704","citing_title":"PCB-QA: Evaluating LLMs over the First Printed Circuit Board Design Question-Answer Dataset","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19220","citing_title":"Position: Uncertainty Quantification in LLMs is Just Unsupervised Clustering","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19118","citing_title":"DP-FlogTinyLLM: Differentially private federated log anomaly detection using Tiny LLMs","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FQ32LH2OFET6F7WFBT36C7DVV6","json":"https://pith.science/pith/FQ32LH2OFET6F7WFBT36C7DVV6.json","graph_json":"https://pith.science/api/pith-number/FQ32LH2OFET6F7WFBT36C7DVV6/graph.json","events_json":"https://pith.science/api/pith-number/FQ32LH2OFET6F7WFBT36C7DVV6/events.json","paper":"https://pith.science/paper/FQ32LH2O"},"agent_actions":{"view_html":"https://pith.science/pith/FQ32LH2OFET6F7WFBT36C7DVV6","download_json":"https://pith.science/pith/FQ32LH2OFET6F7WFBT36C7DVV6.json","view_paper":"https://pith.science/paper/FQ32LH2O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.09135&json=true","fetch_graph":"https://pith.science/api/pith-number/FQ32LH2OFET6F7WFBT36C7DVV6/graph.json","fetch_events":"https://pith.science/api/pith-number/FQ32LH2OFET6F7WFBT36C7DVV6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FQ32LH2OFET6F7WFBT36C7DVV6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FQ32LH2OFET6F7WFBT36C7DVV6/action/storage_attestation","attest_author":"https://pith.science/pith/FQ32LH2OFET6F7WFBT36C7DVV6/action/author_attestation","sign_citation":"https://pith.science/pith/FQ32LH2OFET6F7WFBT36C7DVV6/action/citation_signature","submit_replication":"https://pith.science/pith/FQ32LH2OFET6F7WFBT36C7DVV6/action/replication_record"}},"created_at":"2026-07-05T08:07:45.576082+00:00","updated_at":"2026-07-05T08:07:45.576082+00:00"}