{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5QHGO3NTWWVP4N2UOGCATFHULQ","short_pith_number":"pith:5QHGO3NT","schema_version":"1.0","canonical_sha256":"ec0e676db3b5aafe375471840994f45c020cb87a429351f79edbad8c2f50d3b0","source":{"kind":"arxiv","id":"2305.09246","version":1},"attestation_state":"computed","paper":{"title":"Maybe Only 0.5% Data is Needed: A Preliminary Exploration of Low Training Data Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Hantao Yang, Hao Chen, Junbo Zhao, Qi Zhang, Xiaomeng Hu, Xuetao Ma, Yifan Yanggong, Yiming Zhang","submitted_at":"2023-05-16T07:52:57Z","abstract_excerpt":"Instruction tuning for large language models (LLMs) has gained attention from researchers due to its ability to unlock the potential of LLMs in following instructions. While instruction tuning offers advantages for facilitating the adaptation of large language models (LLMs) to downstream tasks as a fine-tuning approach, training models with tens of millions or even billions of parameters on large amounts of data results in unaffordable computational costs. To address this, we focus on reducing the data used in LLM instruction tuning to decrease training costs and improve data efficiency, dubbe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.09246","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-05-16T07:52:57Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"810de6cb18560f221bcfb811ed5f617d1b7078d29192252ed4fd767948a06b90","abstract_canon_sha256":"47244b136252dceff0ccf293e542e63c771d5c64e2c3d88529752aa6d37f9476"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:10:45.550360Z","signature_b64":"LKR4isn+O0hh2XgPbA601OKldVFhnleeck/vFrJiHCIBirGnQwCrsa6iuCp43pPjgZNbI1H5mH2BW39lxraCDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec0e676db3b5aafe375471840994f45c020cb87a429351f79edbad8c2f50d3b0","last_reissued_at":"2026-07-05T06:10:45.549839Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:10:45.549839Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Maybe Only 0.5% Data is Needed: A Preliminary Exploration of Low Training Data Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Hantao Yang, Hao Chen, Junbo Zhao, Qi Zhang, Xiaomeng Hu, Xuetao Ma, Yifan Yanggong, Yiming Zhang","submitted_at":"2023-05-16T07:52:57Z","abstract_excerpt":"Instruction tuning for large language models (LLMs) has gained attention from researchers due to its ability to unlock the potential of LLMs in following instructions. While instruction tuning offers advantages for facilitating the adaptation of large language models (LLMs) to downstream tasks as a fine-tuning approach, training models with tens of millions or even billions of parameters on large amounts of data results in unaffordable computational costs. To address this, we focus on reducing the data used in LLM instruction tuning to decrease training costs and improve data efficiency, dubbe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.09246","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.09246/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.09246","created_at":"2026-07-05T06:10:45.549902+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.09246v1","created_at":"2026-07-05T06:10:45.549902+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.09246","created_at":"2026-07-05T06:10:45.549902+00:00"},{"alias_kind":"pith_short_12","alias_value":"5QHGO3NTWWVP","created_at":"2026-07-05T06:10:45.549902+00:00"},{"alias_kind":"pith_short_16","alias_value":"5QHGO3NTWWVP4N2U","created_at":"2026-07-05T06:10:45.549902+00:00"},{"alias_kind":"pith_short_8","alias_value":"5QHGO3NT","created_at":"2026-07-05T06:10:45.549902+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.20791","citing_title":"From Cool Demos to Production-Ready FMware: Core Challenges and a Technology Roadmap","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":184,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00369","citing_title":"InvEvolve: Evolving White-Box Inventory Policies via Large Language Models with Performance Guarantees","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00369","citing_title":"InvEvolve: Evolving White-Box Inventory Policies via Large Language Models with Performance Guarantees","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11810","citing_title":"GRACE: A Dynamic Coreset Selection Framework for Large Language Model Optimization","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17650","citing_title":"Measuring Distribution Shift in User Prompts and Its Effects on LLM Performance","ref_index":71,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5QHGO3NTWWVP4N2UOGCATFHULQ","json":"https://pith.science/pith/5QHGO3NTWWVP4N2UOGCATFHULQ.json","graph_json":"https://pith.science/api/pith-number/5QHGO3NTWWVP4N2UOGCATFHULQ/graph.json","events_json":"https://pith.science/api/pith-number/5QHGO3NTWWVP4N2UOGCATFHULQ/events.json","paper":"https://pith.science/paper/5QHGO3NT"},"agent_actions":{"view_html":"https://pith.science/pith/5QHGO3NTWWVP4N2UOGCATFHULQ","download_json":"https://pith.science/pith/5QHGO3NTWWVP4N2UOGCATFHULQ.json","view_paper":"https://pith.science/paper/5QHGO3NT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.09246&json=true","fetch_graph":"https://pith.science/api/pith-number/5QHGO3NTWWVP4N2UOGCATFHULQ/graph.json","fetch_events":"https://pith.science/api/pith-number/5QHGO3NTWWVP4N2UOGCATFHULQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5QHGO3NTWWVP4N2UOGCATFHULQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5QHGO3NTWWVP4N2UOGCATFHULQ/action/storage_attestation","attest_author":"https://pith.science/pith/5QHGO3NTWWVP4N2UOGCATFHULQ/action/author_attestation","sign_citation":"https://pith.science/pith/5QHGO3NTWWVP4N2UOGCATFHULQ/action/citation_signature","submit_replication":"https://pith.science/pith/5QHGO3NTWWVP4N2UOGCATFHULQ/action/replication_record"}},"created_at":"2026-07-05T06:10:45.549902+00:00","updated_at":"2026-07-05T06:10:45.549902+00:00"}