{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3J6YA33R6KZ2THLXUGASKQS6N5","short_pith_number":"pith:3J6YA33R","schema_version":"1.0","canonical_sha256":"da7d806f71f2b3a99d77a18125425e6f549f98eb49dc08901e9c6883fa69689d","source":{"kind":"arxiv","id":"2305.11383","version":2},"attestation_state":"computed","paper":{"title":"Do Models Really Learn to Follow Instructions? An Empirical Study of Instruction Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Nanyun Peng, Po-Nien Kung","submitted_at":"2023-05-19T02:00:47Z","abstract_excerpt":"Recent works on instruction tuning (IT) have achieved great performance with zero-shot generalizability to unseen tasks. With additional context (e.g., task definition, examples) provided to models for fine-tuning, they achieved much higher performance than untuned models. Despite impressive performance gains, what models learn from IT remains understudied. In this work, we analyze how models utilize instructions during IT by comparing model training with altered vs. original instructions. Specifically, we create simplified task definitions by removing all semantic components and only leaving "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.11383","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-05-19T02:00:47Z","cross_cats_sorted":[],"title_canon_sha256":"c1ac39ce2401f38cc03b7364cf1530f6ad50a92ba3e2fc875d7187b575017415","abstract_canon_sha256":"0aeb6b436e6d07a3972b2f89c533decf5da06361bcc80d88096becd4d029e4ed"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:14:10.734004Z","signature_b64":"2HbUfBK+2BDYSd7YLIHeyfu5JvGDVtrNtlFFVUGXWSoqHpPLNCApZaFEYQr9j0O8wV+nnSD8isDZ0c8dlViWBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"da7d806f71f2b3a99d77a18125425e6f549f98eb49dc08901e9c6883fa69689d","last_reissued_at":"2026-07-05T06:14:10.733593Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:14:10.733593Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Models Really Learn to Follow Instructions? An Empirical Study of Instruction Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Nanyun Peng, Po-Nien Kung","submitted_at":"2023-05-19T02:00:47Z","abstract_excerpt":"Recent works on instruction tuning (IT) have achieved great performance with zero-shot generalizability to unseen tasks. With additional context (e.g., task definition, examples) provided to models for fine-tuning, they achieved much higher performance than untuned models. Despite impressive performance gains, what models learn from IT remains understudied. In this work, we analyze how models utilize instructions during IT by comparing model training with altered vs. original instructions. Specifically, we create simplified task definitions by removing all semantic components and only leaving "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.11383","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.11383/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.11383","created_at":"2026-07-05T06:14:10.733651+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.11383v2","created_at":"2026-07-05T06:14:10.733651+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.11383","created_at":"2026-07-05T06:14:10.733651+00:00"},{"alias_kind":"pith_short_12","alias_value":"3J6YA33R6KZ2","created_at":"2026-07-05T06:14:10.733651+00:00"},{"alias_kind":"pith_short_16","alias_value":"3J6YA33R6KZ2THLX","created_at":"2026-07-05T06:14:10.733651+00:00"},{"alias_kind":"pith_short_8","alias_value":"3J6YA33R","created_at":"2026-07-05T06:14:10.733651+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.21573","citing_title":"Instruction Learning Paradigms: A Dual Perspective on White-box and Black-box LLMs","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3J6YA33R6KZ2THLXUGASKQS6N5","json":"https://pith.science/pith/3J6YA33R6KZ2THLXUGASKQS6N5.json","graph_json":"https://pith.science/api/pith-number/3J6YA33R6KZ2THLXUGASKQS6N5/graph.json","events_json":"https://pith.science/api/pith-number/3J6YA33R6KZ2THLXUGASKQS6N5/events.json","paper":"https://pith.science/paper/3J6YA33R"},"agent_actions":{"view_html":"https://pith.science/pith/3J6YA33R6KZ2THLXUGASKQS6N5","download_json":"https://pith.science/pith/3J6YA33R6KZ2THLXUGASKQS6N5.json","view_paper":"https://pith.science/paper/3J6YA33R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.11383&json=true","fetch_graph":"https://pith.science/api/pith-number/3J6YA33R6KZ2THLXUGASKQS6N5/graph.json","fetch_events":"https://pith.science/api/pith-number/3J6YA33R6KZ2THLXUGASKQS6N5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3J6YA33R6KZ2THLXUGASKQS6N5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3J6YA33R6KZ2THLXUGASKQS6N5/action/storage_attestation","attest_author":"https://pith.science/pith/3J6YA33R6KZ2THLXUGASKQS6N5/action/author_attestation","sign_citation":"https://pith.science/pith/3J6YA33R6KZ2THLXUGASKQS6N5/action/citation_signature","submit_replication":"https://pith.science/pith/3J6YA33R6KZ2THLXUGASKQS6N5/action/replication_record"}},"created_at":"2026-07-05T06:14:10.733651+00:00","updated_at":"2026-07-05T06:14:10.733651+00:00"}