{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XEGI62BJDXCS4IN27UTKZVHY3L","short_pith_number":"pith:XEGI62BJ","schema_version":"1.0","canonical_sha256":"b90c8f68291dc52e21bafd26acd4f8dad900d837e2cb4e847e079bb8344e133d","source":{"kind":"arxiv","id":"2402.11161","version":5},"attestation_state":"computed","paper":{"title":"PEDANTS: Cheap but Effective and Interpretable Answer Equivalence","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Huy Nghiem, Ishani Mondal, Jordan Lee Boyd-Graber, Yijun Liang, Zongxia Li","submitted_at":"2024-02-17T01:56:19Z","abstract_excerpt":"Question answering (QA) can only make progress if we know if an answer is correct, but current answer correctness (AC) metrics struggle with verbose, free-form answers from large language models (LLMs). There are two challenges with current short-form QA evaluations: a lack of diverse styles of evaluation data and an over-reliance on expensive and slow LLMs. LLM-based scorers correlate better with humans, but this expensive task has only been tested on limited QA datasets. We rectify these issues by providing rubrics and datasets for evaluating machine QA adopted from the Trivia community. We "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.11161","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-17T01:56:19Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"aa9e3987242e98b679da36327755e268159ef09e3c9d58c38b4f84b3bbe755ac","abstract_canon_sha256":"b074fbca825ad626ca2b90267e3af47871a0b454dd2eb5d247adceee551d7996"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:26.505257Z","signature_b64":"GpnIq+t6g4Y+JtLT9sFJjmbxMVLO2/nxkgQIni55H2OCeJTqAqCkhTdZn31rn4jsmp7rHscOD+osUd8six4GCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b90c8f68291dc52e21bafd26acd4f8dad900d837e2cb4e847e079bb8344e133d","last_reissued_at":"2026-07-05T09:19:26.504753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:26.504753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PEDANTS: Cheap but Effective and Interpretable Answer Equivalence","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Huy Nghiem, Ishani Mondal, Jordan Lee Boyd-Graber, Yijun Liang, Zongxia Li","submitted_at":"2024-02-17T01:56:19Z","abstract_excerpt":"Question answering (QA) can only make progress if we know if an answer is correct, but current answer correctness (AC) metrics struggle with verbose, free-form answers from large language models (LLMs). There are two challenges with current short-form QA evaluations: a lack of diverse styles of evaluation data and an over-reliance on expensive and slow LLMs. LLM-based scorers correlate better with humans, but this expensive task has only been tested on limited QA datasets. We rectify these issues by providing rubrics and datasets for evaluating machine QA adopted from the Trivia community. We "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.11161","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.11161/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.11161","created_at":"2026-07-05T09:19:26.504814+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.11161v5","created_at":"2026-07-05T09:19:26.504814+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.11161","created_at":"2026-07-05T09:19:26.504814+00:00"},{"alias_kind":"pith_short_12","alias_value":"XEGI62BJDXCS","created_at":"2026-07-05T09:19:26.504814+00:00"},{"alias_kind":"pith_short_16","alias_value":"XEGI62BJDXCS4IN2","created_at":"2026-07-05T09:19:26.504814+00:00"},{"alias_kind":"pith_short_8","alias_value":"XEGI62BJ","created_at":"2026-07-05T09:19:26.504814+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.17196","citing_title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","ref_index":84,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XEGI62BJDXCS4IN27UTKZVHY3L","json":"https://pith.science/pith/XEGI62BJDXCS4IN27UTKZVHY3L.json","graph_json":"https://pith.science/api/pith-number/XEGI62BJDXCS4IN27UTKZVHY3L/graph.json","events_json":"https://pith.science/api/pith-number/XEGI62BJDXCS4IN27UTKZVHY3L/events.json","paper":"https://pith.science/paper/XEGI62BJ"},"agent_actions":{"view_html":"https://pith.science/pith/XEGI62BJDXCS4IN27UTKZVHY3L","download_json":"https://pith.science/pith/XEGI62BJDXCS4IN27UTKZVHY3L.json","view_paper":"https://pith.science/paper/XEGI62BJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.11161&json=true","fetch_graph":"https://pith.science/api/pith-number/XEGI62BJDXCS4IN27UTKZVHY3L/graph.json","fetch_events":"https://pith.science/api/pith-number/XEGI62BJDXCS4IN27UTKZVHY3L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XEGI62BJDXCS4IN27UTKZVHY3L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XEGI62BJDXCS4IN27UTKZVHY3L/action/storage_attestation","attest_author":"https://pith.science/pith/XEGI62BJDXCS4IN27UTKZVHY3L/action/author_attestation","sign_citation":"https://pith.science/pith/XEGI62BJDXCS4IN27UTKZVHY3L/action/citation_signature","submit_replication":"https://pith.science/pith/XEGI62BJDXCS4IN27UTKZVHY3L/action/replication_record"}},"created_at":"2026-07-05T09:19:26.504814+00:00","updated_at":"2026-07-05T09:19:26.504814+00:00"}