{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:H22SXQZ25ZDYUYUNVDGSKEAZCA","short_pith_number":"pith:H22SXQZ2","schema_version":"1.0","canonical_sha256":"3eb52bc33aee478a628da8cd251019101cac43b7d257a7488be1da34a5058b74","source":{"kind":"arxiv","id":"2412.15683","version":1},"attestation_state":"computed","paper":{"title":"Variability Need Not Imply Error: The Case of Adequate but Semantically Distinct Responses","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Evgenia Ilia, Wilker Aziz","submitted_at":"2024-12-20T09:02:26Z","abstract_excerpt":"With the broader use of language models (LMs) comes the need to estimate their ability to respond reliably to prompts (e.g., are generated responses likely to be correct?). Uncertainty quantification tools (notions of confidence and entropy, i.a.) can be used to that end (e.g., to reject a response when the model is `uncertain'). For example, Kuhn et al. (semantic entropy; 2022b) regard semantic variation amongst sampled responses as evidence that the model `struggles' with the prompt and that the LM is likely to err. We argue that semantic variability need not imply error--this being especial"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15683","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-20T09:02:26Z","cross_cats_sorted":[],"title_canon_sha256":"f5b5787fb3288cb195757799a639382f88fe1989be4ad5329fbffec08fe62c36","abstract_canon_sha256":"1068d327a486241a91e7e95aa2b945f13e5b38d3545ce09fca33087208b94e70"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:52:28.251822Z","signature_b64":"07AO6dXEhMLNelxhj4Mz5D3/XqlwFCTYrTa9/yC2qd7Pie+SdepSbAmlbN+A0r5QtKBZAKbvor6bLYBtETJ8Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3eb52bc33aee478a628da8cd251019101cac43b7d257a7488be1da34a5058b74","last_reissued_at":"2026-07-05T09:52:28.251402Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:52:28.251402Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Variability Need Not Imply Error: The Case of Adequate but Semantically Distinct Responses","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Evgenia Ilia, Wilker Aziz","submitted_at":"2024-12-20T09:02:26Z","abstract_excerpt":"With the broader use of language models (LMs) comes the need to estimate their ability to respond reliably to prompts (e.g., are generated responses likely to be correct?). Uncertainty quantification tools (notions of confidence and entropy, i.a.) can be used to that end (e.g., to reject a response when the model is `uncertain'). For example, Kuhn et al. (semantic entropy; 2022b) regard semantic variation amongst sampled responses as evidence that the model `struggles' with the prompt and that the LM is likely to err. We argue that semantic variability need not imply error--this being especial"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15683","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15683/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15683","created_at":"2026-07-05T09:52:28.251459+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15683v1","created_at":"2026-07-05T09:52:28.251459+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15683","created_at":"2026-07-05T09:52:28.251459+00:00"},{"alias_kind":"pith_short_12","alias_value":"H22SXQZ25ZDY","created_at":"2026-07-05T09:52:28.251459+00:00"},{"alias_kind":"pith_short_16","alias_value":"H22SXQZ25ZDYUYUN","created_at":"2026-07-05T09:52:28.251459+00:00"},{"alias_kind":"pith_short_8","alias_value":"H22SXQZ2","created_at":"2026-07-05T09:52:28.251459+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H22SXQZ25ZDYUYUNVDGSKEAZCA","json":"https://pith.science/pith/H22SXQZ25ZDYUYUNVDGSKEAZCA.json","graph_json":"https://pith.science/api/pith-number/H22SXQZ25ZDYUYUNVDGSKEAZCA/graph.json","events_json":"https://pith.science/api/pith-number/H22SXQZ25ZDYUYUNVDGSKEAZCA/events.json","paper":"https://pith.science/paper/H22SXQZ2"},"agent_actions":{"view_html":"https://pith.science/pith/H22SXQZ25ZDYUYUNVDGSKEAZCA","download_json":"https://pith.science/pith/H22SXQZ25ZDYUYUNVDGSKEAZCA.json","view_paper":"https://pith.science/paper/H22SXQZ2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15683&json=true","fetch_graph":"https://pith.science/api/pith-number/H22SXQZ25ZDYUYUNVDGSKEAZCA/graph.json","fetch_events":"https://pith.science/api/pith-number/H22SXQZ25ZDYUYUNVDGSKEAZCA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H22SXQZ25ZDYUYUNVDGSKEAZCA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H22SXQZ25ZDYUYUNVDGSKEAZCA/action/storage_attestation","attest_author":"https://pith.science/pith/H22SXQZ25ZDYUYUNVDGSKEAZCA/action/author_attestation","sign_citation":"https://pith.science/pith/H22SXQZ25ZDYUYUNVDGSKEAZCA/action/citation_signature","submit_replication":"https://pith.science/pith/H22SXQZ25ZDYUYUNVDGSKEAZCA/action/replication_record"}},"created_at":"2026-07-05T09:52:28.251459+00:00","updated_at":"2026-07-05T09:52:28.251459+00:00"}