{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:XVPP7YOEN7KGWCFD5WTRRB7KQM","short_pith_number":"pith:XVPP7YOE","canonical_record":{"source":{"id":"2210.11416","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-20T16:58:32Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"ac4d8df0f72b5148980e394515a5066a99f2e64d36f41debaa0e16ab4d467e35","abstract_canon_sha256":"87e5face6580403fceec34a77ce7a85a49c0bff2fd3d5d515fa90bf7cb6ef944"},"schema_version":"1.0"},"canonical_sha256":"bd5effe1c46fd46b08a3eda71887ea830cee9d3f9c1aafe52d6186c5933dc785","source":{"kind":"arxiv","id":"2210.11416","version":5},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.11416","created_at":"2026-07-05T05:23:12Z"},{"alias_kind":"arxiv_version","alias_value":"2210.11416v5","created_at":"2026-07-05T05:23:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.11416","created_at":"2026-07-05T05:23:12Z"},{"alias_kind":"pith_short_12","alias_value":"XVPP7YOEN7KG","created_at":"2026-07-05T05:23:12Z"},{"alias_kind":"pith_short_16","alias_value":"XVPP7YOEN7KGWCFD","created_at":"2026-07-05T05:23:12Z"},{"alias_kind":"pith_short_8","alias_value":"XVPP7YOE","created_at":"2026-07-05T05:23:12Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:XVPP7YOEN7KGWCFD5WTRRB7KQM","target":"record","payload":{"canonical_record":{"source":{"id":"2210.11416","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-20T16:58:32Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"ac4d8df0f72b5148980e394515a5066a99f2e64d36f41debaa0e16ab4d467e35","abstract_canon_sha256":"87e5face6580403fceec34a77ce7a85a49c0bff2fd3d5d515fa90bf7cb6ef944"},"schema_version":"1.0"},"canonical_sha256":"bd5effe1c46fd46b08a3eda71887ea830cee9d3f9c1aafe52d6186c5933dc785","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:23:12.562132Z","signature_b64":"VjnxpawaoYM8i22QBdIu5ok4Gk4Aq/QUVQ8GrQqRVgfvldARBtNw45LaCKkIERqt+0k59K4Mufk1K1ZL8u9qAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bd5effe1c46fd46b08a3eda71887ea830cee9d3f9c1aafe52d6186c5933dc785","last_reissued_at":"2026-07-05T05:23:12.561575Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:23:12.561575Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2210.11416","source_version":5,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:23:12Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"hE1JKB4C8gSi1tJ7bzHPtX7B1P+T/KO8+MN42rRkUFYR+j3OsGmAjoiD3NWRU6q++ri3+HcrFu8GnV/cj5/4Dg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-23T18:18:36.808199Z"},"content_sha256":"7ebe0da2ff15566091334603b524bc95bc5db7c3dd6cd4a55e521eb6d143b0ed","schema_version":"1.0","event_id":"sha256:7ebe0da2ff15566091334603b524bc95bc5db7c3dd6cd4a55e521eb6d143b0ed"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:XVPP7YOEN7KGWCFD5WTRRB7KQM","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Scaling Instruction-Finetuned Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Instruction finetuning on 1.8K tasks plus chain-of-thought data lifts PaLM 540B performance by 9.4 percent on average.","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Aakanksha Chowdhery, Adam Roberts, Adams Yu, Albert Webson, Alex Castro-Ros, Andrew Dai, Barret Zoph, Dasha Valter, Denny Zhou, Ed H. Chi, Gaurav Mishra, Hongkun Yu, Hyung Won Chung, Jacob Devlin, Jason Wei, Jeff Dean, Kevin Robinson, Le Hou, Marie Pellat, Mirac Suzgun, Mostafa Dehghani, Quoc V. Le, Sharan Narang, Shayne Longpre, Shixiang Shane Gu, Siddhartha Brahma, Slav Petrov, Vincent Zhao, William Fedus, Xinyun Chen, Xuezhi Wang, Yanping Huang, Yi Tay, Yunxuan Li, Zhuyun Dai","submitted_at":"2022-10-20T16:58:32Z","abstract_excerpt":"Finetuning language models on a collection of datasets phrased as instructions has been shown to improve model performance and generalization to unseen tasks. In this paper we explore instruction finetuning with a particular focus on (1) scaling the number of tasks, (2) scaling the model size, and (3) finetuning on chain-of-thought data. We find that instruction finetuning with the above aspects dramatically improves performance on a variety of model classes (PaLM, T5, U-PaLM), prompting setups (zero-shot, few-shot, CoT), and evaluation benchmarks (MMLU, BBH, TyDiQA, MGSM, open-ended generatio"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Flan-PaLM 540B instruction-finetuned on 1.8K tasks outperforms PALM 540B by a large margin (+9.4% on average). Flan-PaLM 540B achieves state-of-the-art performance on several benchmarks, such as 75.2% on five-shot MMLU.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the particular collection of 1.8K tasks and the chosen evaluation benchmarks are sufficiently representative of the space of possible instructions and real-world use cases so that the observed gains will generalize.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Instruction finetuning on 1.8K tasks with scaled model size and CoT data improves language model performance, with Flan-PaLM 540B outperforming base PaLM by 9.4% on average and reaching 75.2% on five-shot MMLU.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Instruction finetuning on 1.8K tasks plus chain-of-thought data lifts PaLM 540B performance by 9.4 percent on average.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"4077ec884314705ea86b5cd8130c0d4a30f8fe3348a9f3b33b2d11ff3e2f8e08"},"source":{"id":"2210.11416","kind":"arxiv","version":5},"verdict":{"id":"4fe1970d-e5b8-4043-b571-5ef5371a5be4","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-12T01:17:46.327451Z","strongest_claim":"Flan-PaLM 540B instruction-finetuned on 1.8K tasks outperforms PALM 540B by a large margin (+9.4% on average). Flan-PaLM 540B achieves state-of-the-art performance on several benchmarks, such as 75.2% on five-shot MMLU.","one_line_summary":"Instruction finetuning on 1.8K tasks with scaled model size and CoT data improves language model performance, with Flan-PaLM 540B outperforming base PaLM by 9.4% on average and reaching 75.2% on five-shot MMLU.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the particular collection of 1.8K tasks and the chosen evaluation benchmarks are sufficiently representative of the space of possible instructions and real-world use cases so that the observed gains will generalize.","pith_extraction_headline":"Instruction finetuning on 1.8K tasks plus chain-of-thought data lifts PaLM 540B performance by 9.4 percent on average."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.11416/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":2,"snapshot_sha256":"41fb733359f9673f7f0e17838057cbf81ae0c8d6f7cdbf2a256a5df8e287f8dc"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"4fe1970d-e5b8-4043-b571-5ef5371a5be4"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:23:12Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2XT93U4KX8WHfvnXkFxQqSEDZP4R/H3L47Udb5ayoVP2kSmNb6U3K5YBINp7KULKT7JoesFu1I35LLEHCdxpAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-23T18:18:36.808681Z"},"content_sha256":"6a671d45e39149f797b66cb49aaaac05aeda293eee0df38b19a204e02176e779","schema_version":"1.0","event_id":"sha256:6a671d45e39149f797b66cb49aaaac05aeda293eee0df38b19a204e02176e779"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/XVPP7YOEN7KGWCFD5WTRRB7KQM/bundle.json","state_url":"https://pith.science/pith/XVPP7YOEN7KGWCFD5WTRRB7KQM/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/XVPP7YOEN7KGWCFD5WTRRB7KQM/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-23T18:18:36Z","links":{"resolver":"https://pith.science/pith/XVPP7YOEN7KGWCFD5WTRRB7KQM","bundle":"https://pith.science/pith/XVPP7YOEN7KGWCFD5WTRRB7KQM/bundle.json","state":"https://pith.science/pith/XVPP7YOEN7KGWCFD5WTRRB7KQM/state.json","well_known_bundle":"https://pith.science/.well-known/pith/XVPP7YOEN7KGWCFD5WTRRB7KQM/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:XVPP7YOEN7KGWCFD5WTRRB7KQM","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"87e5face6580403fceec34a77ce7a85a49c0bff2fd3d5d515fa90bf7cb6ef944","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-20T16:58:32Z","title_canon_sha256":"ac4d8df0f72b5148980e394515a5066a99f2e64d36f41debaa0e16ab4d467e35"},"schema_version":"1.0","source":{"id":"2210.11416","kind":"arxiv","version":5}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.11416","created_at":"2026-07-05T05:23:12Z"},{"alias_kind":"arxiv_version","alias_value":"2210.11416v5","created_at":"2026-07-05T05:23:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.11416","created_at":"2026-07-05T05:23:12Z"},{"alias_kind":"pith_short_12","alias_value":"XVPP7YOEN7KG","created_at":"2026-07-05T05:23:12Z"},{"alias_kind":"pith_short_16","alias_value":"XVPP7YOEN7KGWCFD","created_at":"2026-07-05T05:23:12Z"},{"alias_kind":"pith_short_8","alias_value":"XVPP7YOE","created_at":"2026-07-05T05:23:12Z"}],"graph_snapshots":[{"event_id":"sha256:6a671d45e39149f797b66cb49aaaac05aeda293eee0df38b19a204e02176e779","target":"graph","created_at":"2026-07-05T05:23:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Flan-PaLM 540B instruction-finetuned on 1.8K tasks outperforms PALM 540B by a large margin (+9.4% on average). Flan-PaLM 540B achieves state-of-the-art performance on several benchmarks, such as 75.2% on five-shot MMLU."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the particular collection of 1.8K tasks and the chosen evaluation benchmarks are sufficiently representative of the space of possible instructions and real-world use cases so that the observed gains will generalize."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Instruction finetuning on 1.8K tasks with scaled model size and CoT data improves language model performance, with Flan-PaLM 540B outperforming base PaLM by 9.4% on average and reaching 75.2% on five-shot MMLU."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Instruction finetuning on 1.8K tasks plus chain-of-thought data lifts PaLM 540B performance by 9.4 percent on average."}],"snapshot_sha256":"4077ec884314705ea86b5cd8130c0d4a30f8fe3348a9f3b33b2d11ff3e2f8e08"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"41fb733359f9673f7f0e17838057cbf81ae0c8d6f7cdbf2a256a5df8e287f8dc"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2210.11416/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Finetuning language models on a collection of datasets phrased as instructions has been shown to improve model performance and generalization to unseen tasks. In this paper we explore instruction finetuning with a particular focus on (1) scaling the number of tasks, (2) scaling the model size, and (3) finetuning on chain-of-thought data. We find that instruction finetuning with the above aspects dramatically improves performance on a variety of model classes (PaLM, T5, U-PaLM), prompting setups (zero-shot, few-shot, CoT), and evaluation benchmarks (MMLU, BBH, TyDiQA, MGSM, open-ended generatio","authors_text":"Aakanksha Chowdhery, Adam Roberts, Adams Yu, Albert Webson, Alex Castro-Ros, Andrew Dai, Barret Zoph, Dasha Valter, Denny Zhou, Ed H. Chi, Gaurav Mishra, Hongkun Yu, Hyung Won Chung, Jacob Devlin, Jason Wei, Jeff Dean, Kevin Robinson, Le Hou, Marie Pellat, Mirac Suzgun, Mostafa Dehghani, Quoc V. Le, Sharan Narang, Shayne Longpre, Shixiang Shane Gu, Siddhartha Brahma, Slav Petrov, Vincent Zhao, William Fedus, Xinyun Chen, Xuezhi Wang, Yanping Huang, Yi Tay, Yunxuan Li, Zhuyun Dai","cross_cats":["cs.CL"],"headline":"Instruction finetuning on 1.8K tasks plus chain-of-thought data lifts PaLM 540B performance by 9.4 percent on average.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-20T16:58:32Z","title":"Scaling Instruction-Finetuned Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.11416","kind":"arxiv","version":5},"verdict":{"created_at":"2026-05-12T01:17:46.327451Z","id":"4fe1970d-e5b8-4043-b571-5ef5371a5be4","model_set":{"reader":"grok-4.3"},"one_line_summary":"Instruction finetuning on 1.8K tasks with scaled model size and CoT data improves language model performance, with Flan-PaLM 540B outperforming base PaLM by 9.4% on average and reaching 75.2% on five-shot MMLU.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Instruction finetuning on 1.8K tasks plus chain-of-thought data lifts PaLM 540B performance by 9.4 percent on average.","strongest_claim":"Flan-PaLM 540B instruction-finetuned on 1.8K tasks outperforms PALM 540B by a large margin (+9.4% on average). Flan-PaLM 540B achieves state-of-the-art performance on several benchmarks, such as 75.2% on five-shot MMLU.","weakest_assumption":"That the particular collection of 1.8K tasks and the chosen evaluation benchmarks are sufficiently representative of the space of possible instructions and real-world use cases so that the observed gains will generalize."}},"verdict_id":"4fe1970d-e5b8-4043-b571-5ef5371a5be4"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:7ebe0da2ff15566091334603b524bc95bc5db7c3dd6cd4a55e521eb6d143b0ed","target":"record","created_at":"2026-07-05T05:23:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"87e5face6580403fceec34a77ce7a85a49c0bff2fd3d5d515fa90bf7cb6ef944","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-20T16:58:32Z","title_canon_sha256":"ac4d8df0f72b5148980e394515a5066a99f2e64d36f41debaa0e16ab4d467e35"},"schema_version":"1.0","source":{"id":"2210.11416","kind":"arxiv","version":5}},"canonical_sha256":"bd5effe1c46fd46b08a3eda71887ea830cee9d3f9c1aafe52d6186c5933dc785","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"bd5effe1c46fd46b08a3eda71887ea830cee9d3f9c1aafe52d6186c5933dc785","first_computed_at":"2026-07-05T05:23:12.561575Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:23:12.561575Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"VjnxpawaoYM8i22QBdIu5ok4Gk4Aq/QUVQ8GrQqRVgfvldARBtNw45LaCKkIERqt+0k59K4Mufk1K1ZL8u9qAw==","signature_status":"signed_v1","signed_at":"2026-07-05T05:23:12.562132Z","signed_message":"canonical_sha256_bytes"},"source_id":"2210.11416","source_kind":"arxiv","source_version":5}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:7ebe0da2ff15566091334603b524bc95bc5db7c3dd6cd4a55e521eb6d143b0ed","sha256:6a671d45e39149f797b66cb49aaaac05aeda293eee0df38b19a204e02176e779"],"state_sha256":"4d5362b86e86d12a3bc9e94fc92722013d3cfc0452a9960a844f7b454d0fd3ff"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"w8HBE4xmTk2ytBgJELu1NK3Bq6tJhV8mlHTL854SsOGMsO7W7LPvGJg+znRSaupQ85K9DjgfXV8xZMqLYIioBw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-23T18:18:36.811561Z","bundle_sha256":"ebf73d76b0bd42b14cffd72bde5b7206a816ec1b6101418bc6843766413c894f"}}