{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:GX5BU5LI6GWPP5JMKOO3JQUCIS","short_pith_number":"pith:GX5BU5LI","canonical_record":{"source":{"id":"2302.04761","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-02-09T16:49:57Z","cross_cats_sorted":[],"title_canon_sha256":"ac89952d2b848a6ac56c753381f13f20da2c5403fdb25857f29b4e148f49b01a","abstract_canon_sha256":"b37fb1bcf250cad2f5e55422bc081001a9f5a4f3e3a18db12f64de95719d7ea3"},"schema_version":"1.0"},"canonical_sha256":"35fa1a7568f1acf7f52c539db4c28244816d1cb31d7e997cc81478733edb938c","source":{"kind":"arxiv","id":"2302.04761","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2302.04761","created_at":"2026-07-05T05:40:18Z"},{"alias_kind":"arxiv_version","alias_value":"2302.04761v1","created_at":"2026-07-05T05:40:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.04761","created_at":"2026-07-05T05:40:18Z"},{"alias_kind":"pith_short_12","alias_value":"GX5BU5LI6GWP","created_at":"2026-07-05T05:40:18Z"},{"alias_kind":"pith_short_16","alias_value":"GX5BU5LI6GWPP5JM","created_at":"2026-07-05T05:40:18Z"},{"alias_kind":"pith_short_8","alias_value":"GX5BU5LI","created_at":"2026-07-05T05:40:18Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:GX5BU5LI6GWPP5JMKOO3JQUCIS","target":"record","payload":{"canonical_record":{"source":{"id":"2302.04761","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-02-09T16:49:57Z","cross_cats_sorted":[],"title_canon_sha256":"ac89952d2b848a6ac56c753381f13f20da2c5403fdb25857f29b4e148f49b01a","abstract_canon_sha256":"b37fb1bcf250cad2f5e55422bc081001a9f5a4f3e3a18db12f64de95719d7ea3"},"schema_version":"1.0"},"canonical_sha256":"35fa1a7568f1acf7f52c539db4c28244816d1cb31d7e997cc81478733edb938c","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:40:18.200405Z","signature_b64":"N7HK91unnIZ8tZHEYXj53a5LddPI2s2QP+DpS7ZnPxgavm24SLcQP3rqXKdxElzLtXT0CQhu14MqK8yCAfGqAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"35fa1a7568f1acf7f52c539db4c28244816d1cb31d7e997cc81478733edb938c","last_reissued_at":"2026-07-05T05:40:18.200018Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:40:18.200018Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2302.04761","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:40:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4G8n6ZxIJnYKRspHOMI3dOQENlGD6LunrRSaiPok3becwYcJjqmoSsh28Me3R7qBWXMKUsDtSQl4hIPeUV5YAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T09:15:09.615493Z"},"content_sha256":"465fb698a11a992ee7b601d42ed4b55667449d8f79d16277eae91c166483cf4e","schema_version":"1.0","event_id":"sha256:465fb698a11a992ee7b601d42ed4b55667449d8f79d16277eae91c166483cf4e"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:GX5BU5LI6GWPP5JMKOO3JQUCIS","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Toolformer: Language Models Can Teach Themselves to Use Tools","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Language models can teach themselves to use external tools via APIs, improving zero-shot performance on tasks like arithmetic and factual lookup without losing core language abilities.","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jane Dwivedi-Yu, Luke Zettlemoyer, Maria Lomeli, Nicola Cancedda, Roberta Raileanu, Roberto Dess\\`i, Thomas Scialom, Timo Schick","submitted_at":"2023-02-09T16:49:57Z","abstract_excerpt":"Language models (LMs) exhibit remarkable abilities to solve new tasks from just a few examples or textual instructions, especially at scale. They also, paradoxically, struggle with basic functionality, such as arithmetic or factual lookup, where much simpler and smaller models excel. In this paper, we show that LMs can teach themselves to use external tools via simple APIs and achieve the best of both worlds. We introduce Toolformer, a model trained to decide which APIs to call, when to call them, what arguments to pass, and how to best incorporate the results into future token prediction. Thi"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Toolformer achieves substantially improved zero-shot performance across a variety of downstream tasks, often competitive with much larger models, without sacrificing its core language modeling abilities.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That a small number of demonstrations per API is sufficient for the model to learn reliable decisions about when to call tools, what arguments to pass, and how to incorporate results, without the training process introducing harmful biases or over-reliance on the tools.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Language models can be trained in a self-supervised way to decide when and how to use external tools via APIs, leading to better zero-shot performance on downstream tasks without losing core language modeling ability.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Language models can teach themselves to use external tools via APIs, improving zero-shot performance on tasks like arithmetic and factual lookup without losing core language abilities.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"176759590509e609f2fac5fcd6ed417b6f500409b15a3f60681cd470207060e2"},"source":{"id":"2302.04761","kind":"arxiv","version":1},"verdict":{"id":"6ffada89-a108-48c6-82f9-b9f9ae4cd097","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T19:51:28.053365Z","strongest_claim":"Toolformer achieves substantially improved zero-shot performance across a variety of downstream tasks, often competitive with much larger models, without sacrificing its core language modeling abilities.","one_line_summary":"Language models can be trained in a self-supervised way to decide when and how to use external tools via APIs, leading to better zero-shot performance on downstream tasks without losing core language modeling ability.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That a small number of demonstrations per API is sufficient for the model to learn reliable decisions about when to call tools, what arguments to pass, and how to incorporate results, without the training process introducing harmful biases or over-reliance on the tools.","pith_extraction_headline":"Language models can teach themselves to use external tools via APIs, improving zero-shot performance on tasks like arithmetic and factual lookup without losing core language abilities."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.04761/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":2,"snapshot_sha256":"e4079b30a09b5ee74faf5358d1ba67d63d64d95e83144fb72f0e504afe0f6024"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"6ffada89-a108-48c6-82f9-b9f9ae4cd097"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:40:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xHmfeaYaNEmPW8ey+S7/p1WCrcvWF+7smRvwNBbcgJ/jwswXOINhWL9/4LHiNfbQURtcPeR6xrNOc8C7MNC7Dg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T09:15:09.616882Z"},"content_sha256":"86fa9df7fc830be3e3ea2ac0debfb5f8db54967eea67110e77aaac1adb3b4c0b","schema_version":"1.0","event_id":"sha256:86fa9df7fc830be3e3ea2ac0debfb5f8db54967eea67110e77aaac1adb3b4c0b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/GX5BU5LI6GWPP5JMKOO3JQUCIS/bundle.json","state_url":"https://pith.science/pith/GX5BU5LI6GWPP5JMKOO3JQUCIS/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/GX5BU5LI6GWPP5JMKOO3JQUCIS/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-11T09:15:09Z","links":{"resolver":"https://pith.science/pith/GX5BU5LI6GWPP5JMKOO3JQUCIS","bundle":"https://pith.science/pith/GX5BU5LI6GWPP5JMKOO3JQUCIS/bundle.json","state":"https://pith.science/pith/GX5BU5LI6GWPP5JMKOO3JQUCIS/state.json","well_known_bundle":"https://pith.science/.well-known/pith/GX5BU5LI6GWPP5JMKOO3JQUCIS/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:GX5BU5LI6GWPP5JMKOO3JQUCIS","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b37fb1bcf250cad2f5e55422bc081001a9f5a4f3e3a18db12f64de95719d7ea3","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-02-09T16:49:57Z","title_canon_sha256":"ac89952d2b848a6ac56c753381f13f20da2c5403fdb25857f29b4e148f49b01a"},"schema_version":"1.0","source":{"id":"2302.04761","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2302.04761","created_at":"2026-07-05T05:40:18Z"},{"alias_kind":"arxiv_version","alias_value":"2302.04761v1","created_at":"2026-07-05T05:40:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.04761","created_at":"2026-07-05T05:40:18Z"},{"alias_kind":"pith_short_12","alias_value":"GX5BU5LI6GWP","created_at":"2026-07-05T05:40:18Z"},{"alias_kind":"pith_short_16","alias_value":"GX5BU5LI6GWPP5JM","created_at":"2026-07-05T05:40:18Z"},{"alias_kind":"pith_short_8","alias_value":"GX5BU5LI","created_at":"2026-07-05T05:40:18Z"}],"graph_snapshots":[{"event_id":"sha256:86fa9df7fc830be3e3ea2ac0debfb5f8db54967eea67110e77aaac1adb3b4c0b","target":"graph","created_at":"2026-07-05T05:40:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Toolformer achieves substantially improved zero-shot performance across a variety of downstream tasks, often competitive with much larger models, without sacrificing its core language modeling abilities."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That a small number of demonstrations per API is sufficient for the model to learn reliable decisions about when to call tools, what arguments to pass, and how to incorporate results, without the training process introducing harmful biases or over-reliance on the tools."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Language models can be trained in a self-supervised way to decide when and how to use external tools via APIs, leading to better zero-shot performance on downstream tasks without losing core language modeling ability."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Language models can teach themselves to use external tools via APIs, improving zero-shot performance on tasks like arithmetic and factual lookup without losing core language abilities."}],"snapshot_sha256":"176759590509e609f2fac5fcd6ed417b6f500409b15a3f60681cd470207060e2"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"e4079b30a09b5ee74faf5358d1ba67d63d64d95e83144fb72f0e504afe0f6024"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2302.04761/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Language models (LMs) exhibit remarkable abilities to solve new tasks from just a few examples or textual instructions, especially at scale. They also, paradoxically, struggle with basic functionality, such as arithmetic or factual lookup, where much simpler and smaller models excel. In this paper, we show that LMs can teach themselves to use external tools via simple APIs and achieve the best of both worlds. We introduce Toolformer, a model trained to decide which APIs to call, when to call them, what arguments to pass, and how to best incorporate the results into future token prediction. Thi","authors_text":"Jane Dwivedi-Yu, Luke Zettlemoyer, Maria Lomeli, Nicola Cancedda, Roberta Raileanu, Roberto Dess\\`i, Thomas Scialom, Timo Schick","cross_cats":[],"headline":"Language models can teach themselves to use external tools via APIs, improving zero-shot performance on tasks like arithmetic and factual lookup without losing core language abilities.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-02-09T16:49:57Z","title":"Toolformer: Language Models Can Teach Themselves to Use Tools"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.04761","kind":"arxiv","version":1},"verdict":{"created_at":"2026-05-10T19:51:28.053365Z","id":"6ffada89-a108-48c6-82f9-b9f9ae4cd097","model_set":{"reader":"grok-4.3"},"one_line_summary":"Language models can be trained in a self-supervised way to decide when and how to use external tools via APIs, leading to better zero-shot performance on downstream tasks without losing core language modeling ability.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Language models can teach themselves to use external tools via APIs, improving zero-shot performance on tasks like arithmetic and factual lookup without losing core language abilities.","strongest_claim":"Toolformer achieves substantially improved zero-shot performance across a variety of downstream tasks, often competitive with much larger models, without sacrificing its core language modeling abilities.","weakest_assumption":"That a small number of demonstrations per API is sufficient for the model to learn reliable decisions about when to call tools, what arguments to pass, and how to incorporate results, without the training process introducing harmful biases or over-reliance on the tools."}},"verdict_id":"6ffada89-a108-48c6-82f9-b9f9ae4cd097"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:465fb698a11a992ee7b601d42ed4b55667449d8f79d16277eae91c166483cf4e","target":"record","created_at":"2026-07-05T05:40:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b37fb1bcf250cad2f5e55422bc081001a9f5a4f3e3a18db12f64de95719d7ea3","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-02-09T16:49:57Z","title_canon_sha256":"ac89952d2b848a6ac56c753381f13f20da2c5403fdb25857f29b4e148f49b01a"},"schema_version":"1.0","source":{"id":"2302.04761","kind":"arxiv","version":1}},"canonical_sha256":"35fa1a7568f1acf7f52c539db4c28244816d1cb31d7e997cc81478733edb938c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"35fa1a7568f1acf7f52c539db4c28244816d1cb31d7e997cc81478733edb938c","first_computed_at":"2026-07-05T05:40:18.200018Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:40:18.200018Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"N7HK91unnIZ8tZHEYXj53a5LddPI2s2QP+DpS7ZnPxgavm24SLcQP3rqXKdxElzLtXT0CQhu14MqK8yCAfGqAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T05:40:18.200405Z","signed_message":"canonical_sha256_bytes"},"source_id":"2302.04761","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:465fb698a11a992ee7b601d42ed4b55667449d8f79d16277eae91c166483cf4e","sha256:86fa9df7fc830be3e3ea2ac0debfb5f8db54967eea67110e77aaac1adb3b4c0b"],"state_sha256":"68e3d00564295d6961090aac668f156193adf3964f10a23e770e097f9944862b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"EP2DFPXL59QEbL7suXcaTcFR6I1xgdyv2BA3kcRBtqV8oPU9lmnKnbI4VgpMD9jfi9ySU+p3cYTL9//uX97gBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-11T09:15:09.625764Z","bundle_sha256":"fa4c6c5b289af085957069dc55dfed1f12e3e0ead12d48b49ba763d3efb0400a"}}