{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2PUF4QLMC2SXHUPZVQEN7ZF7A7","short_pith_number":"pith:2PUF4QLM","schema_version":"1.0","canonical_sha256":"d3e85e416c16a573d1f9ac08dfe4bf07d10a4d957f9d71ca39098877985d5231","source":{"kind":"arxiv","id":"2501.09254","version":1},"attestation_state":"computed","paper":{"title":"Clone-Robust AI Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GT"],"primary_cat":"cs.LG","authors_text":"Ariel D. Procaccia, Benjamin Schiffer, Shirley Zhang","submitted_at":"2025-01-16T02:43:44Z","abstract_excerpt":"A key challenge in training Large Language Models (LLMs) is properly aligning them with human preferences. Reinforcement Learning with Human Feedback (RLHF) uses pairwise comparisons from human annotators to train reward functions and has emerged as a popular alignment method. However, input datasets in RLHF are not necessarily balanced in the types of questions and answers that are included. Therefore, we want RLHF algorithms to perform well even when the set of alternatives is not uniformly distributed. Drawing on insights from social choice theory, we introduce robustness to approximate clo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.09254","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-01-16T02:43:44Z","cross_cats_sorted":["cs.AI","cs.GT"],"title_canon_sha256":"8201d92f9f7f24cab9950172d56dba2d9fa424969fe249674ab02e0f9ae4487b","abstract_canon_sha256":"d1eef0e327cc13909cb6052363abd3b77d16e99d579095ff4500e8195b8c26ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:01:44.247330Z","signature_b64":"/rSiHl5xH44nDI5LKtQu8hUXf9gpFi/naHH4RjHdmQN6vZnPT+VHWnyJrmiYQlcNsC04c4LSS+eRHZZapZlOAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d3e85e416c16a573d1f9ac08dfe4bf07d10a4d957f9d71ca39098877985d5231","last_reissued_at":"2026-07-05T10:01:44.246870Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:01:44.246870Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Clone-Robust AI Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GT"],"primary_cat":"cs.LG","authors_text":"Ariel D. Procaccia, Benjamin Schiffer, Shirley Zhang","submitted_at":"2025-01-16T02:43:44Z","abstract_excerpt":"A key challenge in training Large Language Models (LLMs) is properly aligning them with human preferences. Reinforcement Learning with Human Feedback (RLHF) uses pairwise comparisons from human annotators to train reward functions and has emerged as a popular alignment method. However, input datasets in RLHF are not necessarily balanced in the types of questions and answers that are included. Therefore, we want RLHF algorithms to perform well even when the set of alternatives is not uniformly distributed. Drawing on insights from social choice theory, we introduce robustness to approximate clo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.09254","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.09254/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.09254","created_at":"2026-07-05T10:01:44.246938+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.09254v1","created_at":"2026-07-05T10:01:44.246938+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.09254","created_at":"2026-07-05T10:01:44.246938+00:00"},{"alias_kind":"pith_short_12","alias_value":"2PUF4QLMC2SX","created_at":"2026-07-05T10:01:44.246938+00:00"},{"alias_kind":"pith_short_16","alias_value":"2PUF4QLMC2SXHUPZ","created_at":"2026-07-05T10:01:44.246938+00:00"},{"alias_kind":"pith_short_8","alias_value":"2PUF4QLM","created_at":"2026-07-05T10:01:44.246938+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.23749","citing_title":"Distortion of AI Alignment: Does Preference Optimization Optimize for Preferences?","ref_index":42,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2PUF4QLMC2SXHUPZVQEN7ZF7A7","json":"https://pith.science/pith/2PUF4QLMC2SXHUPZVQEN7ZF7A7.json","graph_json":"https://pith.science/api/pith-number/2PUF4QLMC2SXHUPZVQEN7ZF7A7/graph.json","events_json":"https://pith.science/api/pith-number/2PUF4QLMC2SXHUPZVQEN7ZF7A7/events.json","paper":"https://pith.science/paper/2PUF4QLM"},"agent_actions":{"view_html":"https://pith.science/pith/2PUF4QLMC2SXHUPZVQEN7ZF7A7","download_json":"https://pith.science/pith/2PUF4QLMC2SXHUPZVQEN7ZF7A7.json","view_paper":"https://pith.science/paper/2PUF4QLM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.09254&json=true","fetch_graph":"https://pith.science/api/pith-number/2PUF4QLMC2SXHUPZVQEN7ZF7A7/graph.json","fetch_events":"https://pith.science/api/pith-number/2PUF4QLMC2SXHUPZVQEN7ZF7A7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2PUF4QLMC2SXHUPZVQEN7ZF7A7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2PUF4QLMC2SXHUPZVQEN7ZF7A7/action/storage_attestation","attest_author":"https://pith.science/pith/2PUF4QLMC2SXHUPZVQEN7ZF7A7/action/author_attestation","sign_citation":"https://pith.science/pith/2PUF4QLMC2SXHUPZVQEN7ZF7A7/action/citation_signature","submit_replication":"https://pith.science/pith/2PUF4QLMC2SXHUPZVQEN7ZF7A7/action/replication_record"}},"created_at":"2026-07-05T10:01:44.246938+00:00","updated_at":"2026-07-05T10:01:44.246938+00:00"}