{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OIJWZ6OMX4SPF2QPSWHUFWZSPZ","short_pith_number":"pith:OIJWZ6OM","schema_version":"1.0","canonical_sha256":"72136cf9ccbf24f2ea0f958f42db327e5b23c016db9f0f6f031bec8f6269988a","source":{"kind":"arxiv","id":"2405.12063","version":2},"attestation_state":"computed","paper":{"title":"CLAMBER: A Benchmark of Identifying and Clarifying Ambiguous Information Needs in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chen Huang, Dingnan Jin, Hongru Liang, Junhong Liu, Peixin Qin, Tat-Seng Chua, Tong Zhang, Wenqiang Lei, Yang Deng","submitted_at":"2024-05-20T14:34:01Z","abstract_excerpt":"Large language models (LLMs) are increasingly used to meet user information needs, but their effectiveness in dealing with user queries that contain various types of ambiguity remains unknown, ultimately risking user trust and satisfaction. To this end, we introduce CLAMBER, a benchmark for evaluating LLMs using a well-organized taxonomy. Building upon the taxonomy, we construct ~12K high-quality data to assess the strengths, weaknesses, and potential risks of various off-the-shelf LLMs. Our findings indicate the limited practical utility of current LLMs in identifying and clarifying ambiguous"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.12063","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-20T14:34:01Z","cross_cats_sorted":[],"title_canon_sha256":"52e06a8260eca5c3793ce27588dc82a4cc5ef2728b3d72273e947c3fefef3207","abstract_canon_sha256":"579b3c8aa27e882b16965ed417bac6840348cbd9fbe8126cbdffbaaff3ad41b7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:54.305214Z","signature_b64":"zAU14Qu9Ha/DyfLK+JNKi5lvIKo8f21gwKp4FbRxNKtRzFl2/msbBhUP/XudAumYx/FQvo6/JEfojDUrPZheBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"72136cf9ccbf24f2ea0f958f42db327e5b23c016db9f0f6f031bec8f6269988a","last_reissued_at":"2026-07-05T08:25:54.304758Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:54.304758Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLAMBER: A Benchmark of Identifying and Clarifying Ambiguous Information Needs in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chen Huang, Dingnan Jin, Hongru Liang, Junhong Liu, Peixin Qin, Tat-Seng Chua, Tong Zhang, Wenqiang Lei, Yang Deng","submitted_at":"2024-05-20T14:34:01Z","abstract_excerpt":"Large language models (LLMs) are increasingly used to meet user information needs, but their effectiveness in dealing with user queries that contain various types of ambiguity remains unknown, ultimately risking user trust and satisfaction. To this end, we introduce CLAMBER, a benchmark for evaluating LLMs using a well-organized taxonomy. Building upon the taxonomy, we construct ~12K high-quality data to assess the strengths, weaknesses, and potential risks of various off-the-shelf LLMs. Our findings indicate the limited practical utility of current LLMs in identifying and clarifying ambiguous"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.12063","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.12063/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.12063","created_at":"2026-07-05T08:25:54.304821+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.12063v2","created_at":"2026-07-05T08:25:54.304821+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.12063","created_at":"2026-07-05T08:25:54.304821+00:00"},{"alias_kind":"pith_short_12","alias_value":"OIJWZ6OMX4SP","created_at":"2026-07-05T08:25:54.304821+00:00"},{"alias_kind":"pith_short_16","alias_value":"OIJWZ6OMX4SPF2QP","created_at":"2026-07-05T08:25:54.304821+00:00"},{"alias_kind":"pith_short_8","alias_value":"OIJWZ6OM","created_at":"2026-07-05T08:25:54.304821+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27209","citing_title":"Learning to Act under Noise: Enhancing Agent Robustness via Noisy Environments","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21296","citing_title":"Discriminatory Compliance: How LLMs Answer Queries from Protected Groups","ref_index":75,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OIJWZ6OMX4SPF2QPSWHUFWZSPZ","json":"https://pith.science/pith/OIJWZ6OMX4SPF2QPSWHUFWZSPZ.json","graph_json":"https://pith.science/api/pith-number/OIJWZ6OMX4SPF2QPSWHUFWZSPZ/graph.json","events_json":"https://pith.science/api/pith-number/OIJWZ6OMX4SPF2QPSWHUFWZSPZ/events.json","paper":"https://pith.science/paper/OIJWZ6OM"},"agent_actions":{"view_html":"https://pith.science/pith/OIJWZ6OMX4SPF2QPSWHUFWZSPZ","download_json":"https://pith.science/pith/OIJWZ6OMX4SPF2QPSWHUFWZSPZ.json","view_paper":"https://pith.science/paper/OIJWZ6OM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.12063&json=true","fetch_graph":"https://pith.science/api/pith-number/OIJWZ6OMX4SPF2QPSWHUFWZSPZ/graph.json","fetch_events":"https://pith.science/api/pith-number/OIJWZ6OMX4SPF2QPSWHUFWZSPZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OIJWZ6OMX4SPF2QPSWHUFWZSPZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OIJWZ6OMX4SPF2QPSWHUFWZSPZ/action/storage_attestation","attest_author":"https://pith.science/pith/OIJWZ6OMX4SPF2QPSWHUFWZSPZ/action/author_attestation","sign_citation":"https://pith.science/pith/OIJWZ6OMX4SPF2QPSWHUFWZSPZ/action/citation_signature","submit_replication":"https://pith.science/pith/OIJWZ6OMX4SPF2QPSWHUFWZSPZ/action/replication_record"}},"created_at":"2026-07-05T08:25:54.304821+00:00","updated_at":"2026-07-05T08:25:54.304821+00:00"}