{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HBDTEZKNNIKLRIYCGZHRVT3B7E","short_pith_number":"pith:HBDTEZKN","schema_version":"1.0","canonical_sha256":"384732654d6a14b8a302364f1acf61f916ed5396edbd00cac7a44317deaaf178","source":{"kind":"arxiv","id":"2406.10323","version":1},"attestation_state":"computed","paper":{"title":"GenQA: Generating Millions of Instructions from a Handful of Prompts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiuhai Chen, John Kirchenbauer, Neel Jain, Rifaa Qadri, Tianyi Zhou, Tom Goldstein, Yuxin Wen","submitted_at":"2024-06-14T17:44:08Z","abstract_excerpt":"Most public instruction finetuning datasets are relatively small compared to the closed source datasets used to train industry models. To study questions about finetuning at scale, such as curricula and learning rate cooldown schedules, there is a need for industrial-scale datasets. However, this scale necessitates a data generation process that is almost entirely automated. In this work, we study methods for generating large instruction datasets from a single prompt. With little human oversight, we get LLMs to write diverse sets of instruction examples ranging from simple completion tasks to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.10323","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-14T17:44:08Z","cross_cats_sorted":[],"title_canon_sha256":"7f7067ab245848fd7299bdecd58843a736b878b37d91c7dfd8c097909a1fb774","abstract_canon_sha256":"f9d480af3f04be2061241fc61b1aaf84dbdbf062a1e359a685eba4d826126786"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:14.939465Z","signature_b64":"mITq+FS9a3xwGTpydHiTRW+7eeElaG444QDFYy8fgPZ5HL9R5Azy32KGkwuo9KBHwfuym6gmrtr5F1VDJTuoCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"384732654d6a14b8a302364f1acf61f916ed5396edbd00cac7a44317deaaf178","last_reissued_at":"2026-07-05T08:32:14.938965Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:14.938965Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GenQA: Generating Millions of Instructions from a Handful of Prompts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiuhai Chen, John Kirchenbauer, Neel Jain, Rifaa Qadri, Tianyi Zhou, Tom Goldstein, Yuxin Wen","submitted_at":"2024-06-14T17:44:08Z","abstract_excerpt":"Most public instruction finetuning datasets are relatively small compared to the closed source datasets used to train industry models. To study questions about finetuning at scale, such as curricula and learning rate cooldown schedules, there is a need for industrial-scale datasets. However, this scale necessitates a data generation process that is almost entirely automated. In this work, we study methods for generating large instruction datasets from a single prompt. With little human oversight, we get LLMs to write diverse sets of instruction examples ranging from simple completion tasks to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.10323","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.10323/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.10323","created_at":"2026-07-05T08:32:14.939024+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.10323v1","created_at":"2026-07-05T08:32:14.939024+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.10323","created_at":"2026-07-05T08:32:14.939024+00:00"},{"alias_kind":"pith_short_12","alias_value":"HBDTEZKNNIKL","created_at":"2026-07-05T08:32:14.939024+00:00"},{"alias_kind":"pith_short_16","alias_value":"HBDTEZKNNIKLRIYC","created_at":"2026-07-05T08:32:14.939024+00:00"},{"alias_kind":"pith_short_8","alias_value":"HBDTEZKN","created_at":"2026-07-05T08:32:14.939024+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18817","citing_title":"Multi-Token Residual Prediction","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18817","citing_title":"Multi-Token Residual Prediction","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2601.09448","citing_title":"One Prompt, Many Sounds: Modeling Listener Variability in LLM-Based Equalization","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21503","citing_title":"MAR: Efficient Large Language Models via Module-aware Architecture Refinement","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2406.08464","citing_title":"Magpie: Alignment Data Synthesis from Scratch by Prompting Aligned LLMs with Nothing","ref_index":95,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HBDTEZKNNIKLRIYCGZHRVT3B7E","json":"https://pith.science/pith/HBDTEZKNNIKLRIYCGZHRVT3B7E.json","graph_json":"https://pith.science/api/pith-number/HBDTEZKNNIKLRIYCGZHRVT3B7E/graph.json","events_json":"https://pith.science/api/pith-number/HBDTEZKNNIKLRIYCGZHRVT3B7E/events.json","paper":"https://pith.science/paper/HBDTEZKN"},"agent_actions":{"view_html":"https://pith.science/pith/HBDTEZKNNIKLRIYCGZHRVT3B7E","download_json":"https://pith.science/pith/HBDTEZKNNIKLRIYCGZHRVT3B7E.json","view_paper":"https://pith.science/paper/HBDTEZKN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.10323&json=true","fetch_graph":"https://pith.science/api/pith-number/HBDTEZKNNIKLRIYCGZHRVT3B7E/graph.json","fetch_events":"https://pith.science/api/pith-number/HBDTEZKNNIKLRIYCGZHRVT3B7E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HBDTEZKNNIKLRIYCGZHRVT3B7E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HBDTEZKNNIKLRIYCGZHRVT3B7E/action/storage_attestation","attest_author":"https://pith.science/pith/HBDTEZKNNIKLRIYCGZHRVT3B7E/action/author_attestation","sign_citation":"https://pith.science/pith/HBDTEZKNNIKLRIYCGZHRVT3B7E/action/citation_signature","submit_replication":"https://pith.science/pith/HBDTEZKNNIKLRIYCGZHRVT3B7E/action/replication_record"}},"created_at":"2026-07-05T08:32:14.939024+00:00","updated_at":"2026-07-05T08:32:14.939024+00:00"}