{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:II27FKLZXFYE4QRI2ZCNUICCTE","short_pith_number":"pith:II27FKLZ","schema_version":"1.0","canonical_sha256":"4235f2a979b9704e4228d644da20429935ca73cf55877ba1788bc835908b811e","source":{"kind":"arxiv","id":"2407.12823","version":1},"attestation_state":"computed","paper":{"title":"WTU-EVAL: A Whether-or-Not Tool Usage Evaluation Benchmark for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jian Liu, Jinan Xu, Kang Liu, Kangyun Ning, Xueqiang Lv, Yisong Su, Yuanzhe Zhang","submitted_at":"2024-07-02T12:07:38Z","abstract_excerpt":"Although Large Language Models (LLMs) excel in NLP tasks, they still need external tools to extend their ability. Current research on tool learning with LLMs often assumes mandatory tool use, which does not always align with real-world situations, where the necessity for tools is uncertain, and incorrect or unnecessary use of tools can damage the general abilities of LLMs. Therefore, we propose to explore whether LLMs can discern their ability boundaries and use tools flexibly. We then introduce the Whether-or-not tool usage Evaluation benchmark (WTU-Eval) to assess LLMs with eleven datasets, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.12823","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-02T12:07:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"69eaaa2efe8baaa24b9350937e29ac3b7db354ca377a8ecf8cc24472d8c87b2e","abstract_canon_sha256":"55066416a11ffa838c5cf4630d390ff7349d2f598e62d4921bbdb848b3580c00"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:12.143829Z","signature_b64":"czVi0fFC+sRvECF5q1UbyPacZAQbJ+c9wVyVGvuPiYQHwBI32CCpUnvNXpnDXTRXDpZv3lNq0+uMs3FrndYtBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4235f2a979b9704e4228d644da20429935ca73cf55877ba1788bc835908b811e","last_reissued_at":"2026-07-05T08:45:12.143381Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:12.143381Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WTU-EVAL: A Whether-or-Not Tool Usage Evaluation Benchmark for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jian Liu, Jinan Xu, Kang Liu, Kangyun Ning, Xueqiang Lv, Yisong Su, Yuanzhe Zhang","submitted_at":"2024-07-02T12:07:38Z","abstract_excerpt":"Although Large Language Models (LLMs) excel in NLP tasks, they still need external tools to extend their ability. Current research on tool learning with LLMs often assumes mandatory tool use, which does not always align with real-world situations, where the necessity for tools is uncertain, and incorrect or unnecessary use of tools can damage the general abilities of LLMs. Therefore, we propose to explore whether LLMs can discern their ability boundaries and use tools flexibly. We then introduce the Whether-or-not tool usage Evaluation benchmark (WTU-Eval) to assess LLMs with eleven datasets, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.12823","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.12823/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.12823","created_at":"2026-07-05T08:45:12.143437+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.12823v1","created_at":"2026-07-05T08:45:12.143437+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.12823","created_at":"2026-07-05T08:45:12.143437+00:00"},{"alias_kind":"pith_short_12","alias_value":"II27FKLZXFYE","created_at":"2026-07-05T08:45:12.143437+00:00"},{"alias_kind":"pith_short_16","alias_value":"II27FKLZXFYE4QRI","created_at":"2026-07-05T08:45:12.143437+00:00"},{"alias_kind":"pith_short_8","alias_value":"II27FKLZ","created_at":"2026-07-05T08:45:12.143437+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.19749","citing_title":"The Tool-Overuse Illusion: Why Does LLM Prefer External Tools over Internal Knowledge?","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09544","citing_title":"TIDE-Bench: Task-Aware and Diagnostic Evaluation of Tool-Integrated Reasoning","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/II27FKLZXFYE4QRI2ZCNUICCTE","json":"https://pith.science/pith/II27FKLZXFYE4QRI2ZCNUICCTE.json","graph_json":"https://pith.science/api/pith-number/II27FKLZXFYE4QRI2ZCNUICCTE/graph.json","events_json":"https://pith.science/api/pith-number/II27FKLZXFYE4QRI2ZCNUICCTE/events.json","paper":"https://pith.science/paper/II27FKLZ"},"agent_actions":{"view_html":"https://pith.science/pith/II27FKLZXFYE4QRI2ZCNUICCTE","download_json":"https://pith.science/pith/II27FKLZXFYE4QRI2ZCNUICCTE.json","view_paper":"https://pith.science/paper/II27FKLZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.12823&json=true","fetch_graph":"https://pith.science/api/pith-number/II27FKLZXFYE4QRI2ZCNUICCTE/graph.json","fetch_events":"https://pith.science/api/pith-number/II27FKLZXFYE4QRI2ZCNUICCTE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/II27FKLZXFYE4QRI2ZCNUICCTE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/II27FKLZXFYE4QRI2ZCNUICCTE/action/storage_attestation","attest_author":"https://pith.science/pith/II27FKLZXFYE4QRI2ZCNUICCTE/action/author_attestation","sign_citation":"https://pith.science/pith/II27FKLZXFYE4QRI2ZCNUICCTE/action/citation_signature","submit_replication":"https://pith.science/pith/II27FKLZXFYE4QRI2ZCNUICCTE/action/replication_record"}},"created_at":"2026-07-05T08:45:12.143437+00:00","updated_at":"2026-07-05T08:45:12.143437+00:00"}