{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KIEUJ2E4UHACDK23OILEJAWFNQ","short_pith_number":"pith:KIEUJ2E4","schema_version":"1.0","canonical_sha256":"520944e89ca1c021ab5b72164482c56c2a6f068da3ac21135890b1b8d3cb3683","source":{"kind":"arxiv","id":"2312.15685","version":2},"attestation_state":"computed","paper":{"title":"What Makes Good Data for Alignment? A Comprehensive Study of Automatic Data Selection in Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Junxian He, Keqing He, Weihao Zeng, Wei Liu, Yong Jiang","submitted_at":"2023-12-25T10:29:28Z","abstract_excerpt":"Instruction tuning is a standard technique employed to align large language models to end tasks and user preferences after the initial pretraining phase. Recent research indicates the critical role of data engineering in instruction tuning -- when appropriately selected, only limited data is necessary to achieve superior performance. However, we still lack a principled understanding of what makes good instruction tuning data for alignment, and how we should select data automatically and effectively. In this work, we delve deeply into automatic data selection strategies for alignment. We start "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.15685","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-25T10:29:28Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"d2ec279d17a5feafe3a7a84274d904006f5d0787e380235c40ecb9fc70acc0fd","abstract_canon_sha256":"14f01d58a4ad7f427dd881724893aa44bd12dc9dbaa16629993e4468f978e47a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:08:19.856887Z","signature_b64":"k2kMh6W2jesJKTs/KKILhKc1DvvOP6XpoH5UpXWJnkUYCFHxXQ8hesnq4eBYClCO5Mn9vZlBJSAPeigHLYBaCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"520944e89ca1c021ab5b72164482c56c2a6f068da3ac21135890b1b8d3cb3683","last_reissued_at":"2026-07-05T08:08:19.856441Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:08:19.856441Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Makes Good Data for Alignment? A Comprehensive Study of Automatic Data Selection in Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Junxian He, Keqing He, Weihao Zeng, Wei Liu, Yong Jiang","submitted_at":"2023-12-25T10:29:28Z","abstract_excerpt":"Instruction tuning is a standard technique employed to align large language models to end tasks and user preferences after the initial pretraining phase. Recent research indicates the critical role of data engineering in instruction tuning -- when appropriately selected, only limited data is necessary to achieve superior performance. However, we still lack a principled understanding of what makes good instruction tuning data for alignment, and how we should select data automatically and effectively. In this work, we delve deeply into automatic data selection strategies for alignment. We start "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.15685","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.15685/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.15685","created_at":"2026-07-05T08:08:19.856499+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.15685v2","created_at":"2026-07-05T08:08:19.856499+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.15685","created_at":"2026-07-05T08:08:19.856499+00:00"},{"alias_kind":"pith_short_12","alias_value":"KIEUJ2E4UHAC","created_at":"2026-07-05T08:08:19.856499+00:00"},{"alias_kind":"pith_short_16","alias_value":"KIEUJ2E4UHACDK23","created_at":"2026-07-05T08:08:19.856499+00:00"},{"alias_kind":"pith_short_8","alias_value":"KIEUJ2E4","created_at":"2026-07-05T08:08:19.856499+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18286","citing_title":"CODEBLOCK: Learning to Supervise Code at the Right Granularity","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07690","citing_title":"HARP: Efficient Data Selection for Finetuning Large Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2403.04652","citing_title":"Yi: Open Foundation Models by 01.AI","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27763","citing_title":"Intent2Tx: Benchmarking LLMs for Translating Natural Language Intents into Ethereum Transactions","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09404","citing_title":"Let the Target Select for Itself: Data Selection via Target-Aligned Paths","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2502.16982","citing_title":"Muon is Scalable for LLM Training","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06230","citing_title":"Safactory: A Scalable Agentic Infrastructure for Training Trustworthy Autonomous Intelligence","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20933","citing_title":"IRIS: Interpolative R\\'enyi Iterative Self-play for Large Language Model Fine-Tuning","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08519","citing_title":"Cram Less to Fit More: Training Data Pruning Improves Memorization of Facts","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06230","citing_title":"Safactory: A Scalable Agentic Infrastructure for Training Trustworthy Autonomous Intelligence","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07381","citing_title":"Escaping the Diversity Trap in Robotic Manipulation via Anchor-Centric Adaptation","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05227","citing_title":"Rethinking Data Curation in LLM Training: Online Reweighting Offers Better Generalization than Offline Methods","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KIEUJ2E4UHACDK23OILEJAWFNQ","json":"https://pith.science/pith/KIEUJ2E4UHACDK23OILEJAWFNQ.json","graph_json":"https://pith.science/api/pith-number/KIEUJ2E4UHACDK23OILEJAWFNQ/graph.json","events_json":"https://pith.science/api/pith-number/KIEUJ2E4UHACDK23OILEJAWFNQ/events.json","paper":"https://pith.science/paper/KIEUJ2E4"},"agent_actions":{"view_html":"https://pith.science/pith/KIEUJ2E4UHACDK23OILEJAWFNQ","download_json":"https://pith.science/pith/KIEUJ2E4UHACDK23OILEJAWFNQ.json","view_paper":"https://pith.science/paper/KIEUJ2E4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.15685&json=true","fetch_graph":"https://pith.science/api/pith-number/KIEUJ2E4UHACDK23OILEJAWFNQ/graph.json","fetch_events":"https://pith.science/api/pith-number/KIEUJ2E4UHACDK23OILEJAWFNQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KIEUJ2E4UHACDK23OILEJAWFNQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KIEUJ2E4UHACDK23OILEJAWFNQ/action/storage_attestation","attest_author":"https://pith.science/pith/KIEUJ2E4UHACDK23OILEJAWFNQ/action/author_attestation","sign_citation":"https://pith.science/pith/KIEUJ2E4UHACDK23OILEJAWFNQ/action/citation_signature","submit_replication":"https://pith.science/pith/KIEUJ2E4UHACDK23OILEJAWFNQ/action/replication_record"}},"created_at":"2026-07-05T08:08:19.856499+00:00","updated_at":"2026-07-05T08:08:19.856499+00:00"}