{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NJAB2J34BUYFNPFRO5GFEVIZBJ","short_pith_number":"pith:NJAB2J34","schema_version":"1.0","canonical_sha256":"6a401d277c0d3056bcb1774c5255190a66133826d5f2f91b1ff8b7abd98d61b7","source":{"kind":"arxiv","id":"2409.06097","version":2},"attestation_state":"computed","paper":{"title":"ClarQ-LLM: A Benchmark for Models Clarifying and Requesting Information in Task-Oriented Dialog","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Changling Li, Jinxia Xie, Luou Wen, Massimo Poesio, Matthew Purver, Yujian Gan","submitted_at":"2024-09-09T22:29:35Z","abstract_excerpt":"We introduce ClarQ-LLM, an evaluation framework consisting of bilingual English-Chinese conversation tasks, conversational agents and evaluation metrics, designed to serve as a strong benchmark for assessing agents' ability to ask clarification questions in task-oriented dialogues. The benchmark includes 31 different task types, each with 10 unique dialogue scenarios between information seeker and provider agents. The scenarios require the seeker to ask questions to resolve uncertainty and gather necessary information to complete tasks. Unlike traditional benchmarks that evaluate agents based "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.06097","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-09T22:29:35Z","cross_cats_sorted":[],"title_canon_sha256":"194c063943fb2d0d0945bb0e3175cc2a958358b26357ce6d0e42a823ea23c069","abstract_canon_sha256":"175d554322b88c7afb809e84c0c271e82e8ba7e1c3d72008c186844828881a87"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:07:03.428652Z","signature_b64":"pzlqYQY6xfaX3dG1OmEkZQNIJ8j+vv/vuLdQA0W+17gNru4yaPdwqauiwSbbrADgh4lbmyWd7vRDHLGACamtDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a401d277c0d3056bcb1774c5255190a66133826d5f2f91b1ff8b7abd98d61b7","last_reissued_at":"2026-07-05T09:07:03.428206Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:07:03.428206Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ClarQ-LLM: A Benchmark for Models Clarifying and Requesting Information in Task-Oriented Dialog","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Changling Li, Jinxia Xie, Luou Wen, Massimo Poesio, Matthew Purver, Yujian Gan","submitted_at":"2024-09-09T22:29:35Z","abstract_excerpt":"We introduce ClarQ-LLM, an evaluation framework consisting of bilingual English-Chinese conversation tasks, conversational agents and evaluation metrics, designed to serve as a strong benchmark for assessing agents' ability to ask clarification questions in task-oriented dialogues. The benchmark includes 31 different task types, each with 10 unique dialogue scenarios between information seeker and provider agents. The scenarios require the seeker to ask questions to resolve uncertainty and gather necessary information to complete tasks. Unlike traditional benchmarks that evaluate agents based "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.06097","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.06097/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.06097","created_at":"2026-07-05T09:07:03.428262+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.06097v2","created_at":"2026-07-05T09:07:03.428262+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.06097","created_at":"2026-07-05T09:07:03.428262+00:00"},{"alias_kind":"pith_short_12","alias_value":"NJAB2J34BUYF","created_at":"2026-07-05T09:07:03.428262+00:00"},{"alias_kind":"pith_short_16","alias_value":"NJAB2J34BUYFNPFR","created_at":"2026-07-05T09:07:03.428262+00:00"},{"alias_kind":"pith_short_8","alias_value":"NJAB2J34","created_at":"2026-07-05T09:07:03.428262+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01182","citing_title":"CA-BED: Conversation-Aware Bayesian Experimental Design","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27209","citing_title":"Learning to Act under Noise: Enhancing Agent Robustness via Noisy Environments","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18768","citing_title":"ClinQueryAgent: A Conversational Agent for Population Health Management","ref_index":154,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18630","citing_title":"SCICONVBENCH: Benchmarking LLMs on Multi-Turn Clarification for Task Formulation in Computational Science","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09228","citing_title":"ProactBench: Beyond What The User Asked For","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19656","citing_title":"Pause or Fabricate? Training Language Models for Grounded Reasoning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06717","citing_title":"Agentic Coding Needs Proactivity, Not Just Autonomy","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NJAB2J34BUYFNPFRO5GFEVIZBJ","json":"https://pith.science/pith/NJAB2J34BUYFNPFRO5GFEVIZBJ.json","graph_json":"https://pith.science/api/pith-number/NJAB2J34BUYFNPFRO5GFEVIZBJ/graph.json","events_json":"https://pith.science/api/pith-number/NJAB2J34BUYFNPFRO5GFEVIZBJ/events.json","paper":"https://pith.science/paper/NJAB2J34"},"agent_actions":{"view_html":"https://pith.science/pith/NJAB2J34BUYFNPFRO5GFEVIZBJ","download_json":"https://pith.science/pith/NJAB2J34BUYFNPFRO5GFEVIZBJ.json","view_paper":"https://pith.science/paper/NJAB2J34","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.06097&json=true","fetch_graph":"https://pith.science/api/pith-number/NJAB2J34BUYFNPFRO5GFEVIZBJ/graph.json","fetch_events":"https://pith.science/api/pith-number/NJAB2J34BUYFNPFRO5GFEVIZBJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NJAB2J34BUYFNPFRO5GFEVIZBJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NJAB2J34BUYFNPFRO5GFEVIZBJ/action/storage_attestation","attest_author":"https://pith.science/pith/NJAB2J34BUYFNPFRO5GFEVIZBJ/action/author_attestation","sign_citation":"https://pith.science/pith/NJAB2J34BUYFNPFRO5GFEVIZBJ/action/citation_signature","submit_replication":"https://pith.science/pith/NJAB2J34BUYFNPFRO5GFEVIZBJ/action/replication_record"}},"created_at":"2026-07-05T09:07:03.428262+00:00","updated_at":"2026-07-05T09:07:03.428262+00:00"}