{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QE337I7AONKQAFZ54ABSCDQFG4","short_pith_number":"pith:QE337I7A","schema_version":"1.0","canonical_sha256":"8137bfa3e0735500173de003210e05373a70cf23fe93d0c84fac7a551c29ab89","source":{"kind":"arxiv","id":"2409.00096","version":1},"attestation_state":"computed","paper":{"title":"Non-instructional Fine-tuning: Enabling Instruction-Following Capabilities in Pre-trained Language Models without Instruction-Following Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hung-yi Lee, Juncheng Xie, Shensian Syu","submitted_at":"2024-08-27T01:21:53Z","abstract_excerpt":"Instruction fine-tuning is crucial for today's large language models (LLMs) to learn to follow instructions and align with human preferences. Conventionally, supervised data, including the instruction and the correct response, is required for instruction fine-tuning. To obtain such data, some researchers prompted well-trained models like GPT-4 to generate instructions and correct responses. In this paper, we propose a novel approach that uses the first half of a random text from OpenWebText as the instruction and GPT-3.5-turbo or GPT-4-turbo to complete the text as the response. Despite the da"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.00096","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-08-27T01:21:53Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"622899d425cb9d0c973213ca5731205184b314a517eb6eabe5a5bf41dc442f9f","abstract_canon_sha256":"244946da91aa933019739bb1b3c0ab0a3af6412078ec73d6107d9fa9bbe3a24d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:01:40.873088Z","signature_b64":"txlsHmBVT4P/c1LCdzhDJI/tjryHTC2zlPIltKy3EDk/P6SUCIU+DXcuGl8F61Vt8enl5VuxyauyD4JSdVMwCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8137bfa3e0735500173de003210e05373a70cf23fe93d0c84fac7a551c29ab89","last_reissued_at":"2026-07-05T09:01:40.872704Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:01:40.872704Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Non-instructional Fine-tuning: Enabling Instruction-Following Capabilities in Pre-trained Language Models without Instruction-Following Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hung-yi Lee, Juncheng Xie, Shensian Syu","submitted_at":"2024-08-27T01:21:53Z","abstract_excerpt":"Instruction fine-tuning is crucial for today's large language models (LLMs) to learn to follow instructions and align with human preferences. Conventionally, supervised data, including the instruction and the correct response, is required for instruction fine-tuning. To obtain such data, some researchers prompted well-trained models like GPT-4 to generate instructions and correct responses. In this paper, we propose a novel approach that uses the first half of a random text from OpenWebText as the instruction and GPT-3.5-turbo or GPT-4-turbo to complete the text as the response. Despite the da"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.00096","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.00096/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.00096","created_at":"2026-07-05T09:01:40.872756+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.00096v1","created_at":"2026-07-05T09:01:40.872756+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.00096","created_at":"2026-07-05T09:01:40.872756+00:00"},{"alias_kind":"pith_short_12","alias_value":"QE337I7AONKQ","created_at":"2026-07-05T09:01:40.872756+00:00"},{"alias_kind":"pith_short_16","alias_value":"QE337I7AONKQAFZ5","created_at":"2026-07-05T09:01:40.872756+00:00"},{"alias_kind":"pith_short_8","alias_value":"QE337I7A","created_at":"2026-07-05T09:01:40.872756+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.07817","citing_title":"On the Effect of Instruction Tuning Loss on Generalization","ref_index":53,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QE337I7AONKQAFZ54ABSCDQFG4","json":"https://pith.science/pith/QE337I7AONKQAFZ54ABSCDQFG4.json","graph_json":"https://pith.science/api/pith-number/QE337I7AONKQAFZ54ABSCDQFG4/graph.json","events_json":"https://pith.science/api/pith-number/QE337I7AONKQAFZ54ABSCDQFG4/events.json","paper":"https://pith.science/paper/QE337I7A"},"agent_actions":{"view_html":"https://pith.science/pith/QE337I7AONKQAFZ54ABSCDQFG4","download_json":"https://pith.science/pith/QE337I7AONKQAFZ54ABSCDQFG4.json","view_paper":"https://pith.science/paper/QE337I7A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.00096&json=true","fetch_graph":"https://pith.science/api/pith-number/QE337I7AONKQAFZ54ABSCDQFG4/graph.json","fetch_events":"https://pith.science/api/pith-number/QE337I7AONKQAFZ54ABSCDQFG4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QE337I7AONKQAFZ54ABSCDQFG4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QE337I7AONKQAFZ54ABSCDQFG4/action/storage_attestation","attest_author":"https://pith.science/pith/QE337I7AONKQAFZ54ABSCDQFG4/action/author_attestation","sign_citation":"https://pith.science/pith/QE337I7AONKQAFZ54ABSCDQFG4/action/citation_signature","submit_replication":"https://pith.science/pith/QE337I7AONKQAFZ54ABSCDQFG4/action/replication_record"}},"created_at":"2026-07-05T09:01:40.872756+00:00","updated_at":"2026-07-05T09:01:40.872756+00:00"}