{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HTFKCUJI3FSNHQCO7CPATW5UGZ","short_pith_number":"pith:HTFKCUJI","schema_version":"1.0","canonical_sha256":"3ccaa15128d964d3c04ef89e09dbb436774b1f076632dd6e7ca1546b2cf3f0ff","source":{"kind":"arxiv","id":"2402.13887","version":2},"attestation_state":"computed","paper":{"title":"Beyond Probabilities: Unveiling the Misalignment in Evaluating Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Chenyang Lyu, Minghao Wu","submitted_at":"2024-02-21T15:58:37Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable capabilities across various applications, fundamentally reshaping the landscape of natural language processing (NLP) research. However, recent evaluation frameworks often rely on the output probabilities of LLMs for predictions, primarily due to computational constraints, diverging from real-world LLM usage scenarios. While widely employed, the efficacy of these probability-based evaluation strategies remains an open research question. This study aims to scrutinize the validity of such probability-based evaluation methods within the con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.13887","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-21T15:58:37Z","cross_cats_sorted":[],"title_canon_sha256":"240bee997740dde93d355e01f93f1a26da867d94a1adaafebbc6582d6c1e525b","abstract_canon_sha256":"144142726bdea7293ad01a5c1253d6988c58ce54480ad88df323dd11240799b9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:41:43.350658Z","signature_b64":"KTVduHE/MrJJ0J2QPKiJ8UJc7SsHQrJyAfUcMN2Gi5HWV0QuzkVu7AvrhPaO8XdrDPsKFfwyoLG8/AqeuUiIAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3ccaa15128d964d3c04ef89e09dbb436774b1f076632dd6e7ca1546b2cf3f0ff","last_reissued_at":"2026-07-05T08:41:43.350253Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:41:43.350253Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Probabilities: Unveiling the Misalignment in Evaluating Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Chenyang Lyu, Minghao Wu","submitted_at":"2024-02-21T15:58:37Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable capabilities across various applications, fundamentally reshaping the landscape of natural language processing (NLP) research. However, recent evaluation frameworks often rely on the output probabilities of LLMs for predictions, primarily due to computational constraints, diverging from real-world LLM usage scenarios. While widely employed, the efficacy of these probability-based evaluation strategies remains an open research question. This study aims to scrutinize the validity of such probability-based evaluation methods within the con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.13887","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.13887/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.13887","created_at":"2026-07-05T08:41:43.350323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.13887v2","created_at":"2026-07-05T08:41:43.350323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.13887","created_at":"2026-07-05T08:41:43.350323+00:00"},{"alias_kind":"pith_short_12","alias_value":"HTFKCUJI3FSN","created_at":"2026-07-05T08:41:43.350323+00:00"},{"alias_kind":"pith_short_16","alias_value":"HTFKCUJI3FSNHQCO","created_at":"2026-07-05T08:41:43.350323+00:00"},{"alias_kind":"pith_short_8","alias_value":"HTFKCUJI","created_at":"2026-07-05T08:41:43.350323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":211,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":210,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HTFKCUJI3FSNHQCO7CPATW5UGZ","json":"https://pith.science/pith/HTFKCUJI3FSNHQCO7CPATW5UGZ.json","graph_json":"https://pith.science/api/pith-number/HTFKCUJI3FSNHQCO7CPATW5UGZ/graph.json","events_json":"https://pith.science/api/pith-number/HTFKCUJI3FSNHQCO7CPATW5UGZ/events.json","paper":"https://pith.science/paper/HTFKCUJI"},"agent_actions":{"view_html":"https://pith.science/pith/HTFKCUJI3FSNHQCO7CPATW5UGZ","download_json":"https://pith.science/pith/HTFKCUJI3FSNHQCO7CPATW5UGZ.json","view_paper":"https://pith.science/paper/HTFKCUJI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.13887&json=true","fetch_graph":"https://pith.science/api/pith-number/HTFKCUJI3FSNHQCO7CPATW5UGZ/graph.json","fetch_events":"https://pith.science/api/pith-number/HTFKCUJI3FSNHQCO7CPATW5UGZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HTFKCUJI3FSNHQCO7CPATW5UGZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HTFKCUJI3FSNHQCO7CPATW5UGZ/action/storage_attestation","attest_author":"https://pith.science/pith/HTFKCUJI3FSNHQCO7CPATW5UGZ/action/author_attestation","sign_citation":"https://pith.science/pith/HTFKCUJI3FSNHQCO7CPATW5UGZ/action/citation_signature","submit_replication":"https://pith.science/pith/HTFKCUJI3FSNHQCO7CPATW5UGZ/action/replication_record"}},"created_at":"2026-07-05T08:41:43.350323+00:00","updated_at":"2026-07-05T08:41:43.350323+00:00"}