{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4KD6IGSSQSCY4PUFTQZDWGB3QN","short_pith_number":"pith:4KD6IGSS","schema_version":"1.0","canonical_sha256":"e287e41a5284858e3e859c323b183b834c6c7be0224c396e906a54b5b84fc87d","source":{"kind":"arxiv","id":"2411.10545","version":2},"attestation_state":"computed","paper":{"title":"Efficient Alignment of Large Language Models via Data Sampling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Amrit Khera, Debojyoti Dutta, Rajat Ghosh","submitted_at":"2024-11-15T19:36:15Z","abstract_excerpt":"LLM alignment ensures that large language models behave safely and effectively by aligning their outputs with human values, goals, and intentions. Aligning LLMs employ huge amounts of data, computation, and time. Moreover, curating data with human feedback is expensive and takes time. Recent research depicts the benefit of data engineering in the fine-tuning and pre-training paradigms to bring down such costs. However, alignment differs from the afore-mentioned paradigms and it is unclear if data efficient alignment is feasible. In this work, we first aim to understand how the performance of L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.10545","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-15T19:36:15Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"555a7b5813627212e196a82a4e3d4c8b6f3a040f3678474de0508af99b79d993","abstract_canon_sha256":"0ea82a6035ae2b2920f32b45ad0a28a1d8cf647a8534b16b168a593ae998a98e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:47.257548Z","signature_b64":"NquMHZRQIDhIl2y0qic1Ar0BNfYGjFXI1BGCPs0akJ8q+dLEo1UFUrZM5W+vVXwqxFhVFHlXcfJLMVDoL8ovBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e287e41a5284858e3e859c323b183b834c6c7be0224c396e906a54b5b84fc87d","last_reissued_at":"2026-07-05T10:15:47.256894Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:47.256894Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Alignment of Large Language Models via Data Sampling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Amrit Khera, Debojyoti Dutta, Rajat Ghosh","submitted_at":"2024-11-15T19:36:15Z","abstract_excerpt":"LLM alignment ensures that large language models behave safely and effectively by aligning their outputs with human values, goals, and intentions. Aligning LLMs employ huge amounts of data, computation, and time. Moreover, curating data with human feedback is expensive and takes time. Recent research depicts the benefit of data engineering in the fine-tuning and pre-training paradigms to bring down such costs. However, alignment differs from the afore-mentioned paradigms and it is unclear if data efficient alignment is feasible. In this work, we first aim to understand how the performance of L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.10545","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.10545/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.10545","created_at":"2026-07-05T10:15:47.256977+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.10545v2","created_at":"2026-07-05T10:15:47.256977+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.10545","created_at":"2026-07-05T10:15:47.256977+00:00"},{"alias_kind":"pith_short_12","alias_value":"4KD6IGSSQSCY","created_at":"2026-07-05T10:15:47.256977+00:00"},{"alias_kind":"pith_short_16","alias_value":"4KD6IGSSQSCY4PUF","created_at":"2026-07-05T10:15:47.256977+00:00"},{"alias_kind":"pith_short_8","alias_value":"4KD6IGSS","created_at":"2026-07-05T10:15:47.256977+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4KD6IGSSQSCY4PUFTQZDWGB3QN","json":"https://pith.science/pith/4KD6IGSSQSCY4PUFTQZDWGB3QN.json","graph_json":"https://pith.science/api/pith-number/4KD6IGSSQSCY4PUFTQZDWGB3QN/graph.json","events_json":"https://pith.science/api/pith-number/4KD6IGSSQSCY4PUFTQZDWGB3QN/events.json","paper":"https://pith.science/paper/4KD6IGSS"},"agent_actions":{"view_html":"https://pith.science/pith/4KD6IGSSQSCY4PUFTQZDWGB3QN","download_json":"https://pith.science/pith/4KD6IGSSQSCY4PUFTQZDWGB3QN.json","view_paper":"https://pith.science/paper/4KD6IGSS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.10545&json=true","fetch_graph":"https://pith.science/api/pith-number/4KD6IGSSQSCY4PUFTQZDWGB3QN/graph.json","fetch_events":"https://pith.science/api/pith-number/4KD6IGSSQSCY4PUFTQZDWGB3QN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4KD6IGSSQSCY4PUFTQZDWGB3QN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4KD6IGSSQSCY4PUFTQZDWGB3QN/action/storage_attestation","attest_author":"https://pith.science/pith/4KD6IGSSQSCY4PUFTQZDWGB3QN/action/author_attestation","sign_citation":"https://pith.science/pith/4KD6IGSSQSCY4PUFTQZDWGB3QN/action/citation_signature","submit_replication":"https://pith.science/pith/4KD6IGSSQSCY4PUFTQZDWGB3QN/action/replication_record"}},"created_at":"2026-07-05T10:15:47.256977+00:00","updated_at":"2026-07-05T10:15:47.256977+00:00"}