{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HSLQWUSVG3RTSMWRSGS6AA24GX","short_pith_number":"pith:HSLQWUSV","schema_version":"1.0","canonical_sha256":"3c970b525536e33932d191a5e0035c35d179d33add895ff63fc2dca92d3b6f73","source":{"kind":"arxiv","id":"2310.03128","version":6},"attestation_state":"computed","paper":{"title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Chenrui Fan, Jiawen Shi, Lichao Sun, Neil Zhenqiang Gong, Pan Zhou, Qihui Zhang, Siyuan Wu, Yao Wan, Yixin Liu, Yuan Li, Yue Huang","submitted_at":"2023-10-04T19:39:26Z","abstract_excerpt":"Large language models (LLMs) have garnered significant attention due to their impressive natural language processing (NLP) capabilities. Recently, many studies have focused on the tool utilization ability of LLMs. They primarily investigated how LLMs effectively collaborate with given specific tools. However, in scenarios where LLMs serve as intelligent agents, as seen in applications like AutoGPT and MetaGPT, LLMs are expected to engage in intricate decision-making processes that involve deciding whether to employ a tool and selecting the most suitable tool(s) from a collection of available t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.03128","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.SE","submitted_at":"2023-10-04T19:39:26Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"c7da8d38263984d5e1809683257ec6a682e539c99f58875aed955cf55f10aeeb","abstract_canon_sha256":"6a6a8b0b7673eaac0189cbd1209e2e069de0bd849970e8223f51c02cc5061362"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:44:32.833761Z","signature_b64":"bfl3nURmqKgLQ0LjUj0x9QU+o7moEPs6ObHHFOpXzNN7m24gWvItbYgNblbW7eGf4GWwpmxTvRt4q32RrZ+wBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c970b525536e33932d191a5e0035c35d179d33add895ff63fc2dca92d3b6f73","last_reissued_at":"2026-07-05T09:44:32.833260Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:44:32.833260Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Chenrui Fan, Jiawen Shi, Lichao Sun, Neil Zhenqiang Gong, Pan Zhou, Qihui Zhang, Siyuan Wu, Yao Wan, Yixin Liu, Yuan Li, Yue Huang","submitted_at":"2023-10-04T19:39:26Z","abstract_excerpt":"Large language models (LLMs) have garnered significant attention due to their impressive natural language processing (NLP) capabilities. Recently, many studies have focused on the tool utilization ability of LLMs. They primarily investigated how LLMs effectively collaborate with given specific tools. However, in scenarios where LLMs serve as intelligent agents, as seen in applications like AutoGPT and MetaGPT, LLMs are expected to engage in intricate decision-making processes that involve deciding whether to employ a tool and selecting the most suitable tool(s) from a collection of available t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.03128","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.03128/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.03128","created_at":"2026-07-05T09:44:32.833320+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.03128v6","created_at":"2026-07-05T09:44:32.833320+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.03128","created_at":"2026-07-05T09:44:32.833320+00:00"},{"alias_kind":"pith_short_12","alias_value":"HSLQWUSVG3RT","created_at":"2026-07-05T09:44:32.833320+00:00"},{"alias_kind":"pith_short_16","alias_value":"HSLQWUSVG3RTSMWR","created_at":"2026-07-05T09:44:32.833320+00:00"},{"alias_kind":"pith_short_8","alias_value":"HSLQWUSV","created_at":"2026-07-05T09:44:32.833320+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18051","citing_title":"Compositional Skill Routing for LLM Agents: Decompose, Retrieve, and Compose","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06566","citing_title":"NTILC: Neural Tool Invocation via Learned Compression","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06667","citing_title":"The Piggyback Hypothesis of Generalization: Explaining and Mitigating Emergent Misalignment","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26154","citing_title":"MemMorph: Tool Hijacking in LLM Agents via Memory Poisoning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02411","citing_title":"FitText: Evolving Agent Tool Ecologies via Memetic Retrieval","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00251","citing_title":"Capability Self-Assessment: Teaching LLMs to Know Their Limits","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22177","citing_title":"Maestro: Reinforcement Learning to Orchestrate Hierarchical Model-Skill Ensembles","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21035","citing_title":"GenoMAS: A Multi-Agent Framework for Scientific Discovery via Code-Driven Gene Expression Analysis","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14038","citing_title":"Model-Adaptive Tool Necessity Reveals the Knowing-Doing Gap in LLM Tool Use","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14892","citing_title":"Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems","ref_index":165,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05561","citing_title":"TrustLLM: Trustworthiness in Large Language Models","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2510.02837","citing_title":"Beyond the Final Answer: Evaluating the Reasoning Trajectories of Tool-Augmented Agents","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2510.23853","citing_title":"Your LLM Agents are Temporally Blind: The Misalignment Between Tool Use Decisions and Human Time Perception","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19793","citing_title":"Prompt Injection Attack to Tool Selection in LLM Agents","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14038","citing_title":"Model-Adaptive Tool Necessity Reveals the Knowing-Doing Gap in LLM Tool Use","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14892","citing_title":"Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems","ref_index":164,"is_internal_anchor":false},{"citing_arxiv_id":"2402.17177","citing_title":"Sora: A Review on Background, Technology, Limitations, and Opportunities of Large Vision Models","ref_index":122,"is_internal_anchor":false},{"citing_arxiv_id":"2506.07982","citing_title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03986","citing_title":"From Intent to Execution: Composing Agentic Workflows with Agent Recommendation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13564","citing_title":"Memory in the Age of AI Agents","ref_index":261,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00737","citing_title":"To Call or Not to Call: A Framework to Assess and Optimize LLM Tool Calling","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2406.12045","citing_title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07251","citing_title":"Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02411","citing_title":"FitText: Evolving Agent Tool Ecologies via Memetic Retrieval","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02682","citing_title":"Hybrid Inspection and Task-Based Access Control in Zero-Trust Agentic AI","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HSLQWUSVG3RTSMWRSGS6AA24GX","json":"https://pith.science/pith/HSLQWUSVG3RTSMWRSGS6AA24GX.json","graph_json":"https://pith.science/api/pith-number/HSLQWUSVG3RTSMWRSGS6AA24GX/graph.json","events_json":"https://pith.science/api/pith-number/HSLQWUSVG3RTSMWRSGS6AA24GX/events.json","paper":"https://pith.science/paper/HSLQWUSV"},"agent_actions":{"view_html":"https://pith.science/pith/HSLQWUSVG3RTSMWRSGS6AA24GX","download_json":"https://pith.science/pith/HSLQWUSVG3RTSMWRSGS6AA24GX.json","view_paper":"https://pith.science/paper/HSLQWUSV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.03128&json=true","fetch_graph":"https://pith.science/api/pith-number/HSLQWUSVG3RTSMWRSGS6AA24GX/graph.json","fetch_events":"https://pith.science/api/pith-number/HSLQWUSVG3RTSMWRSGS6AA24GX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HSLQWUSVG3RTSMWRSGS6AA24GX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HSLQWUSVG3RTSMWRSGS6AA24GX/action/storage_attestation","attest_author":"https://pith.science/pith/HSLQWUSVG3RTSMWRSGS6AA24GX/action/author_attestation","sign_citation":"https://pith.science/pith/HSLQWUSVG3RTSMWRSGS6AA24GX/action/citation_signature","submit_replication":"https://pith.science/pith/HSLQWUSVG3RTSMWRSGS6AA24GX/action/replication_record"}},"created_at":"2026-07-05T09:44:32.833320+00:00","updated_at":"2026-07-05T09:44:32.833320+00:00"}