{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QBHUXFWBFCGOBZCSCN33UHBD4G","short_pith_number":"pith:QBHUXFWB","schema_version":"1.0","canonical_sha256":"804f4b96c1288ce0e4521377ba1c23e19584336b8ad5fe80992108509eef75f4","source":{"kind":"arxiv","id":"2401.17167","version":3},"attestation_state":"computed","paper":{"title":"Planning, Creation, Usage: Benchmarking LLMs for Comprehensive Tool Utilization in Real-World Complex Scenarios","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiahui Gao, Jianqiao Lu, Lifeng Shang, Qi Zhu, Qun Liu, Ruifeng Xu, Shijue Huang, Wanjun Zhong, Weiwen Liu, Xingshan Zeng, Xin Jiang, Yasheng Wang, Yutai Hou","submitted_at":"2024-01-30T16:52:56Z","abstract_excerpt":"The recent trend of using Large Language Models (LLMs) as tool agents in real-world applications underscores the necessity for comprehensive evaluations of their capabilities, particularly in complex scenarios involving planning, creating, and using tools. However, existing benchmarks typically focus on simple synthesized queries that do not reflect real-world complexity, thereby offering limited perspectives in evaluating tool utilization. To address this issue, we present UltraTool, a novel benchmark designed to improve and evaluate LLMs' ability in tool utilization within real-world scenari"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.17167","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-30T16:52:56Z","cross_cats_sorted":[],"title_canon_sha256":"4dbacc2eb562b781a861d1361d9590550a9b39302f81137353710eef71cedd55","abstract_canon_sha256":"5fdf73ac196fbce1c128294ba36e0e8a6e6120fe1dab993ba90a3431ee7eb460"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:26:18.305995Z","signature_b64":"8zIi1R8Jfw1mQhkcOIXsfUaqGXs4IFXigDBrSr/VSzri9O+K35XfZqeqWQcARsJfg2pwQWfypXPG+qP4cslhCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"804f4b96c1288ce0e4521377ba1c23e19584336b8ad5fe80992108509eef75f4","last_reissued_at":"2026-07-05T08:26:18.305380Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:26:18.305380Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Planning, Creation, Usage: Benchmarking LLMs for Comprehensive Tool Utilization in Real-World Complex Scenarios","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiahui Gao, Jianqiao Lu, Lifeng Shang, Qi Zhu, Qun Liu, Ruifeng Xu, Shijue Huang, Wanjun Zhong, Weiwen Liu, Xingshan Zeng, Xin Jiang, Yasheng Wang, Yutai Hou","submitted_at":"2024-01-30T16:52:56Z","abstract_excerpt":"The recent trend of using Large Language Models (LLMs) as tool agents in real-world applications underscores the necessity for comprehensive evaluations of their capabilities, particularly in complex scenarios involving planning, creating, and using tools. However, existing benchmarks typically focus on simple synthesized queries that do not reflect real-world complexity, thereby offering limited perspectives in evaluating tool utilization. To address this issue, we present UltraTool, a novel benchmark designed to improve and evaluate LLMs' ability in tool utilization within real-world scenari"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.17167","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.17167/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.17167","created_at":"2026-07-05T08:26:18.305444+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.17167v3","created_at":"2026-07-05T08:26:18.305444+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.17167","created_at":"2026-07-05T08:26:18.305444+00:00"},{"alias_kind":"pith_short_12","alias_value":"QBHUXFWBFCGO","created_at":"2026-07-05T08:26:18.305444+00:00"},{"alias_kind":"pith_short_16","alias_value":"QBHUXFWBFCGOBZCS","created_at":"2026-07-05T08:26:18.305444+00:00"},{"alias_kind":"pith_short_8","alias_value":"QBHUXFWB","created_at":"2026-07-05T08:26:18.305444+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18636","citing_title":"PEC-Home: Interpretation of Progressively Elliptical Commands in Smart Homes","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2509.18847","citing_title":"Failure Makes the Agent Stronger: Enhancing Accuracy through Structured Reflection for Reliable Tool Interactions","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2503.23278","citing_title":"Model Context Protocol (MCP): Landscape, Security Threats, and Future Research Directions","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QBHUXFWBFCGOBZCSCN33UHBD4G","json":"https://pith.science/pith/QBHUXFWBFCGOBZCSCN33UHBD4G.json","graph_json":"https://pith.science/api/pith-number/QBHUXFWBFCGOBZCSCN33UHBD4G/graph.json","events_json":"https://pith.science/api/pith-number/QBHUXFWBFCGOBZCSCN33UHBD4G/events.json","paper":"https://pith.science/paper/QBHUXFWB"},"agent_actions":{"view_html":"https://pith.science/pith/QBHUXFWBFCGOBZCSCN33UHBD4G","download_json":"https://pith.science/pith/QBHUXFWBFCGOBZCSCN33UHBD4G.json","view_paper":"https://pith.science/paper/QBHUXFWB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.17167&json=true","fetch_graph":"https://pith.science/api/pith-number/QBHUXFWBFCGOBZCSCN33UHBD4G/graph.json","fetch_events":"https://pith.science/api/pith-number/QBHUXFWBFCGOBZCSCN33UHBD4G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QBHUXFWBFCGOBZCSCN33UHBD4G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QBHUXFWBFCGOBZCSCN33UHBD4G/action/storage_attestation","attest_author":"https://pith.science/pith/QBHUXFWBFCGOBZCSCN33UHBD4G/action/author_attestation","sign_citation":"https://pith.science/pith/QBHUXFWBFCGOBZCSCN33UHBD4G/action/citation_signature","submit_replication":"https://pith.science/pith/QBHUXFWBFCGOBZCSCN33UHBD4G/action/replication_record"}},"created_at":"2026-07-05T08:26:18.305444+00:00","updated_at":"2026-07-05T08:26:18.305444+00:00"}