{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HLKOCEWROTA7ZZY4ZY7SRA4X6O","short_pith_number":"pith:HLKOCEWR","schema_version":"1.0","canonical_sha256":"3ad4e112d174c1fce71cce3f288397f396d2a029e21b014ebce3138a5b235d25","source":{"kind":"arxiv","id":"2305.18752","version":1},"attestation_state":"computed","paper":{"title":"GPT4Tools: Teaching Large Language Model to Use Tools via Self-instruction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Lin Song, Rui Yang, Sijie Zhao, Xiu Li, Yanwei Li, Ying Shan, Yixiao Ge","submitted_at":"2023-05-30T05:27:21Z","abstract_excerpt":"This paper aims to efficiently enable Large Language Models (LLMs) to use multimodal tools. Advanced proprietary LLMs, such as ChatGPT and GPT-4, have shown great potential for tool usage through sophisticated prompt engineering. Nevertheless, these models typically rely on prohibitive computational costs and publicly inaccessible data. To address these challenges, we propose the GPT4Tools based on self-instruct to enable open-source LLMs, such as LLaMA and OPT, to use tools. It generates an instruction-following dataset by prompting an advanced teacher with various multi-modal contexts. By us"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.18752","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-05-30T05:27:21Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"253d9d3f7a96e44151d3ab46893e3381afc834c128a5c7ff7fa33d9c074565d2","abstract_canon_sha256":"433f0c9b5b19afc859538737aea27ee92de5255579f628349f8b796c09feb924"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:15:04.481806Z","signature_b64":"MAhUIu5bIsnuXSex2c/6kpM0JBDIzdARdCe9FF2EGLnZmT1pDExepThB+zV75YBdyfhLzdhyn0jvw3i1shW3DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3ad4e112d174c1fce71cce3f288397f396d2a029e21b014ebce3138a5b235d25","last_reissued_at":"2026-07-05T06:15:04.481405Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:15:04.481405Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GPT4Tools: Teaching Large Language Model to Use Tools via Self-instruction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Lin Song, Rui Yang, Sijie Zhao, Xiu Li, Yanwei Li, Ying Shan, Yixiao Ge","submitted_at":"2023-05-30T05:27:21Z","abstract_excerpt":"This paper aims to efficiently enable Large Language Models (LLMs) to use multimodal tools. Advanced proprietary LLMs, such as ChatGPT and GPT-4, have shown great potential for tool usage through sophisticated prompt engineering. Nevertheless, these models typically rely on prohibitive computational costs and publicly inaccessible data. To address these challenges, we propose the GPT4Tools based on self-instruct to enable open-source LLMs, such as LLaMA and OPT, to use tools. It generates an instruction-following dataset by prompting an advanced teacher with various multi-modal contexts. By us"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.18752","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.18752/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.18752","created_at":"2026-07-05T06:15:04.481465+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.18752v1","created_at":"2026-07-05T06:15:04.481465+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.18752","created_at":"2026-07-05T06:15:04.481465+00:00"},{"alias_kind":"pith_short_12","alias_value":"HLKOCEWROTA7","created_at":"2026-07-05T06:15:04.481465+00:00"},{"alias_kind":"pith_short_16","alias_value":"HLKOCEWROTA7ZZY4","created_at":"2026-07-05T06:15:04.481465+00:00"},{"alias_kind":"pith_short_8","alias_value":"HLKOCEWR","created_at":"2026-07-05T06:15:04.481465+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09084","citing_title":"Context-Fractured Decomposition Attacks on Tool-Using LLM Agents: Exploiting Artifact Provenance Gaps","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2403.18814","citing_title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2401.16158","citing_title":"Mobile-Agent: Autonomous Multi-Modal Mobile Device Agent with Visual Perception","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2312.14238","citing_title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","ref_index":164,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03057","citing_title":"Querying Structured Data Through Natural Language Using Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11706","citing_title":"GRAFT: Graph-Tokenized LLMs for Tool Planning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18562","citing_title":"AnchorSeg: Language Grounded Query Banks for Reasoning Segmentation","ref_index":141,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HLKOCEWROTA7ZZY4ZY7SRA4X6O","json":"https://pith.science/pith/HLKOCEWROTA7ZZY4ZY7SRA4X6O.json","graph_json":"https://pith.science/api/pith-number/HLKOCEWROTA7ZZY4ZY7SRA4X6O/graph.json","events_json":"https://pith.science/api/pith-number/HLKOCEWROTA7ZZY4ZY7SRA4X6O/events.json","paper":"https://pith.science/paper/HLKOCEWR"},"agent_actions":{"view_html":"https://pith.science/pith/HLKOCEWROTA7ZZY4ZY7SRA4X6O","download_json":"https://pith.science/pith/HLKOCEWROTA7ZZY4ZY7SRA4X6O.json","view_paper":"https://pith.science/paper/HLKOCEWR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.18752&json=true","fetch_graph":"https://pith.science/api/pith-number/HLKOCEWROTA7ZZY4ZY7SRA4X6O/graph.json","fetch_events":"https://pith.science/api/pith-number/HLKOCEWROTA7ZZY4ZY7SRA4X6O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HLKOCEWROTA7ZZY4ZY7SRA4X6O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HLKOCEWROTA7ZZY4ZY7SRA4X6O/action/storage_attestation","attest_author":"https://pith.science/pith/HLKOCEWROTA7ZZY4ZY7SRA4X6O/action/author_attestation","sign_citation":"https://pith.science/pith/HLKOCEWROTA7ZZY4ZY7SRA4X6O/action/citation_signature","submit_replication":"https://pith.science/pith/HLKOCEWROTA7ZZY4ZY7SRA4X6O/action/replication_record"}},"created_at":"2026-07-05T06:15:04.481465+00:00","updated_at":"2026-07-05T06:15:04.481465+00:00"}